diff --git a/.gitattributes b/.gitattributes
index a6344aac8c09253b3b630fb776ae94478aa0275b..757bf8cf8f1f41236b8c5e9f1691db07111bf76e 100644
--- a/.gitattributes
+++ b/.gitattributes
@@ -33,3 +33,42 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
*.zip filter=lfs diff=lfs merge=lfs -text
*.zst filter=lfs diff=lfs merge=lfs -text
*tfevents* filter=lfs diff=lfs merge=lfs -text
+checkpoint-1000/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-1200/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-1400/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-1600/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-1800/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-200/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-2000/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-2200/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-2400/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-2600/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-2800/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-3000/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-3200/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-3400/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-3600/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-3800/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-400/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-4000/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-4200/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-4400/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-4600/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-4800/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-5000/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-5200/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-5400/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-5600/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-5800/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-600/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-6000/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-6200/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-6400/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-6600/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-6800/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-7000/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-7200/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-7400/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-7460/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+checkpoint-800/tokenizer.json filter=lfs diff=lfs merge=lfs -text
+tokenizer.json filter=lfs diff=lfs merge=lfs -text
diff --git a/README.md b/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/adapter_config.json b/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/adapter_model.safetensors b/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..2d56dd52e04576c533bceb18ec394b50943e9d91
--- /dev/null
+++ b/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e45b98549709b9485f32de70adf0f96edce8ca97b599382c87a97973290a9c87
+size 50360752
diff --git a/chat_template.jinja b/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-1000/README.md b/checkpoint-1000/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-1000/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-1000/adapter_config.json b/checkpoint-1000/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-1000/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-1000/adapter_model.safetensors b/checkpoint-1000/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..e74367298f28c1e41eb82bf2fcb3e7da4b0fc005
--- /dev/null
+++ b/checkpoint-1000/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:4ac3894d523e35894aeae2b1c086e01fcae4ba05069b56b46969eb0f2504c8e6
+size 50360752
diff --git a/checkpoint-1000/chat_template.jinja b/checkpoint-1000/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-1000/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-1000/optimizer.pt b/checkpoint-1000/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..e487d54aaec30206b9956a2e1e8f282392bf3ab4
--- /dev/null
+++ b/checkpoint-1000/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:de1b730d984dc003995b54aff0084253a58ef6b058175c32d84a3d1d1e3c717b
+size 100828235
diff --git a/checkpoint-1000/rng_state.pth b/checkpoint-1000/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..70b420ea263b5924d82b3936b2173962c460358f
--- /dev/null
+++ b/checkpoint-1000/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b8c78882d5b2c84736c1873ea3d414ad53537972adacc7dccf326a9ebc5a0f89
+size 14645
diff --git a/checkpoint-1000/scheduler.pt b/checkpoint-1000/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..bd33870f5c03449172f9c197eca91cf9b1caa083
--- /dev/null
+++ b/checkpoint-1000/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f226462606e1b78f7191acb3109491600d12573ce587511117ddff07f83f50dd
+size 1465
diff --git a/checkpoint-1000/tokenizer.json b/checkpoint-1000/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-1000/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-1000/tokenizer_config.json b/checkpoint-1000/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-1000/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-1000/trainer_state.json b/checkpoint-1000/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..e0823e1b9b5c364710ae8f239051dd22193b9722
--- /dev/null
+++ b/checkpoint-1000/trainer_state.json
@@ -0,0 +1,384 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.26812347085833027,
+ "eval_steps": 500,
+ "global_step": 1000,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 1.2265883151887155e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-1000/training_args.bin b/checkpoint-1000/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-1000/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-1200/README.md b/checkpoint-1200/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-1200/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-1200/adapter_config.json b/checkpoint-1200/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-1200/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-1200/adapter_model.safetensors b/checkpoint-1200/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..c860b0c11b9297970e3b399c4331e9abbff2551f
--- /dev/null
+++ b/checkpoint-1200/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:417989dbb49cc7b8b39ad62db038f967e940507b46ee8c39482d35892346c728
+size 50360752
diff --git a/checkpoint-1200/chat_template.jinja b/checkpoint-1200/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-1200/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-1200/optimizer.pt b/checkpoint-1200/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..55d92805c5ea321f97a89bb4e145e268b5c91858
--- /dev/null
+++ b/checkpoint-1200/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e27e2c9c771096aa76b05aae3b399be1b383b6a9ba510a9615b0f6f8a6f70119
+size 100828235
diff --git a/checkpoint-1200/rng_state.pth b/checkpoint-1200/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..872452357941975936cf2690d425d3e2aa6276ab
--- /dev/null
+++ b/checkpoint-1200/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:96cde9f5973b91631d0681c40dd52cfa938ee26803320fa23d35863f4f5f4217
+size 14645
diff --git a/checkpoint-1200/scheduler.pt b/checkpoint-1200/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..ecf4058648be9b281ad20aa1f50ad6aa383576d8
--- /dev/null
+++ b/checkpoint-1200/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5049e5d5d140d2f9add531f3077ff375aff61236ef16509409553782896408a3
+size 1465
diff --git a/checkpoint-1200/tokenizer.json b/checkpoint-1200/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-1200/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-1200/tokenizer_config.json b/checkpoint-1200/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-1200/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-1200/trainer_state.json b/checkpoint-1200/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..d0041a6ecf2b3675b575a7cdb99dbb3e9a2f11b6
--- /dev/null
+++ b/checkpoint-1200/trainer_state.json
@@ -0,0 +1,454 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.3217481650299963,
+ "eval_steps": 500,
+ "global_step": 1200,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 1.471552740097106e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-1200/training_args.bin b/checkpoint-1200/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-1200/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-1400/README.md b/checkpoint-1400/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-1400/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-1400/adapter_config.json b/checkpoint-1400/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-1400/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-1400/adapter_model.safetensors b/checkpoint-1400/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..6ced79fbab838e8e5473dd33bdb1d8b10f9f8b20
--- /dev/null
+++ b/checkpoint-1400/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:4fe4375c944d5df24e8c793700e13aed770284562c56c3a8cc6808b38bd7abc0
+size 50360752
diff --git a/checkpoint-1400/chat_template.jinja b/checkpoint-1400/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-1400/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-1400/optimizer.pt b/checkpoint-1400/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..4dd9f29db24eb422de368e0698fe783f93f685f8
--- /dev/null
+++ b/checkpoint-1400/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:47d94db2fd46b9422d93e25e91453d20756375ac76aac6663b03b6a6fc25bef4
+size 100828235
diff --git a/checkpoint-1400/rng_state.pth b/checkpoint-1400/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..4fa50e77b3fe1334d29b3eb91d8a99958af49bbe
--- /dev/null
+++ b/checkpoint-1400/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:162580b2a1cba676dfe8710060a053ba2797d5b8bc3c88d23d7586fbc01083eb
+size 14645
diff --git a/checkpoint-1400/scheduler.pt b/checkpoint-1400/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..37b468a8cb502aeadfa2cccf9d4f4784e743fb6e
--- /dev/null
+++ b/checkpoint-1400/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8ae33e797bdb8b83fcfd3d3c8acd9fdd0e0c30ff0058439330b84aaf64f957b7
+size 1465
diff --git a/checkpoint-1400/tokenizer.json b/checkpoint-1400/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-1400/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-1400/tokenizer_config.json b/checkpoint-1400/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-1400/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-1400/trainer_state.json b/checkpoint-1400/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..c0eed54cd0f497731de2e24e0743d9ab1673f23b
--- /dev/null
+++ b/checkpoint-1400/trainer_state.json
@@ -0,0 +1,524 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.37537285920166236,
+ "eval_steps": 500,
+ "global_step": 1400,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 1.716587745411932e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-1400/training_args.bin b/checkpoint-1400/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-1400/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-1600/README.md b/checkpoint-1600/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-1600/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-1600/adapter_config.json b/checkpoint-1600/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-1600/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-1600/adapter_model.safetensors b/checkpoint-1600/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..99e87d799cc5eff48620b823bf22a276f435f79d
--- /dev/null
+++ b/checkpoint-1600/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c3c7d20f9c92c83c3e12ff31d71f58e8745aaf3d1f044cb5705d000d14c20c1a
+size 50360752
diff --git a/checkpoint-1600/chat_template.jinja b/checkpoint-1600/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-1600/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-1600/optimizer.pt b/checkpoint-1600/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..7e6dc0a9433d391d36a05423bd2cce992fd3d7f9
--- /dev/null
+++ b/checkpoint-1600/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ad434d84ce693267a0de2e31b4b3d4720932da875afad04190d6e7a5cdf89efc
+size 100828235
diff --git a/checkpoint-1600/rng_state.pth b/checkpoint-1600/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..0f730a278bfffb68aa4937ec72d8254d55df68b3
--- /dev/null
+++ b/checkpoint-1600/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:219bdda0c11190b268d7e89b412db53f775797054aa6786905e1d43e958bd9ff
+size 14645
diff --git a/checkpoint-1600/scheduler.pt b/checkpoint-1600/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..833d3603ffb66bab9f7e86122acad5c41ac4e14e
--- /dev/null
+++ b/checkpoint-1600/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3b564ed9b760aa1e9458ee373fe09c366ab6ead87c7c8a1b22b2df6345a99f67
+size 1465
diff --git a/checkpoint-1600/tokenizer.json b/checkpoint-1600/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-1600/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-1600/tokenizer_config.json b/checkpoint-1600/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-1600/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-1600/trainer_state.json b/checkpoint-1600/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..c095f74584b4510a4dd075afd7f73771a033f26c
--- /dev/null
+++ b/checkpoint-1600/trainer_state.json
@@ -0,0 +1,594 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.4289975533733284,
+ "eval_steps": 500,
+ "global_step": 1600,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 1.9650753089415782e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-1600/training_args.bin b/checkpoint-1600/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-1600/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-1800/README.md b/checkpoint-1800/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-1800/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-1800/adapter_config.json b/checkpoint-1800/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-1800/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-1800/adapter_model.safetensors b/checkpoint-1800/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..3a2134f3c62b3ee453c505dae936807ad2e190ff
--- /dev/null
+++ b/checkpoint-1800/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ad650210ad416713c19e4dfcf053b5187aad88eb7af7712cd60a8c9b074573d3
+size 50360752
diff --git a/checkpoint-1800/chat_template.jinja b/checkpoint-1800/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-1800/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-1800/optimizer.pt b/checkpoint-1800/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..f4cd6a0c8b36f8b5dc0f89189b45db01a2491349
--- /dev/null
+++ b/checkpoint-1800/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:fa831b9206f20a510557e89c9e2492c185d5825c708533098fd94b41ec3ff9a1
+size 100828235
diff --git a/checkpoint-1800/rng_state.pth b/checkpoint-1800/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..5b08f27b93cc04fc64daab11464d45eeeb2ec5a7
--- /dev/null
+++ b/checkpoint-1800/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3c5db57a6f231ea08d151da49a2a16729dde52229c213948f11f47d94db8d1ba
+size 14645
diff --git a/checkpoint-1800/scheduler.pt b/checkpoint-1800/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..acb28cfdf1d66e698d4e5244644182b1964687d6
--- /dev/null
+++ b/checkpoint-1800/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:abd2865514c1384e677df4f6924dbe82113eabee331604867a3a83cd07c3a494
+size 1465
diff --git a/checkpoint-1800/tokenizer.json b/checkpoint-1800/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-1800/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-1800/tokenizer_config.json b/checkpoint-1800/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-1800/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-1800/trainer_state.json b/checkpoint-1800/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..20009ebbdf33c1c019a321d43b215dc1af773396
--- /dev/null
+++ b/checkpoint-1800/trainer_state.json
@@ -0,0 +1,664 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.48262224754499444,
+ "eval_steps": 500,
+ "global_step": 1800,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 2.2117387050620314e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-1800/training_args.bin b/checkpoint-1800/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-1800/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-200/README.md b/checkpoint-200/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-200/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-200/adapter_config.json b/checkpoint-200/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-200/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-200/adapter_model.safetensors b/checkpoint-200/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..3fdd8483ea567e069f94fc6dd03aa3e3191aef2a
--- /dev/null
+++ b/checkpoint-200/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0d237f34f08ec41369ad8edd17a168795fc3671b94230c59bb06dbbe4d98b9eb
+size 50360752
diff --git a/checkpoint-200/chat_template.jinja b/checkpoint-200/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-200/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-200/optimizer.pt b/checkpoint-200/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..255da496fea8cc23c80e3bcac42407aaadb45519
--- /dev/null
+++ b/checkpoint-200/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:538c31e3f2f0229d802a70a0379ac1ca3aaa6fe67be0d5827c4b8fd6d1f17ed1
+size 100828235
diff --git a/checkpoint-200/rng_state.pth b/checkpoint-200/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..dab0e2f4f62aa0f90488b90a5bfc4f4854401576
--- /dev/null
+++ b/checkpoint-200/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:45dc76167e6efb068117728fc0ca9c4f1c97428a1b876b6cf7fd1d0bb4d198f2
+size 14645
diff --git a/checkpoint-200/scheduler.pt b/checkpoint-200/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..6afb91e10fce82d6a4fc32a1a5ed0a2530936cef
--- /dev/null
+++ b/checkpoint-200/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:01a6f3499d0acb3d598cd8b7037c45927c9a1d5f189a53dc2d71d0774aaba8ef
+size 1465
diff --git a/checkpoint-200/tokenizer.json b/checkpoint-200/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-200/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-200/tokenizer_config.json b/checkpoint-200/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-200/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-200/trainer_state.json b/checkpoint-200/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..6551aeb99d64cc8e4bbc5f3ad079ad4b78218588
--- /dev/null
+++ b/checkpoint-200/trainer_state.json
@@ -0,0 +1,104 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.05362469417166605,
+ "eval_steps": 500,
+ "global_step": 200,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 2.461979015351501e+16,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-200/training_args.bin b/checkpoint-200/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-200/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-2000/README.md b/checkpoint-2000/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-2000/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-2000/adapter_config.json b/checkpoint-2000/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-2000/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-2000/adapter_model.safetensors b/checkpoint-2000/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..00855b76cfc24c5c07c71308e10b64990ad85a36
--- /dev/null
+++ b/checkpoint-2000/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:91c1d9072dd3bc5619ee8538cea97d04eea3180a80350e6f16fc8e1de6d46681
+size 50360752
diff --git a/checkpoint-2000/chat_template.jinja b/checkpoint-2000/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-2000/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-2000/optimizer.pt b/checkpoint-2000/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..d5febde6da50172eaef9714d1893eb233f883970
--- /dev/null
+++ b/checkpoint-2000/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:655f460495b3f9994817435da5aab200c465986092596a6cdbaa165ead1eaf82
+size 100828235
diff --git a/checkpoint-2000/rng_state.pth b/checkpoint-2000/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..15c397e4ae186ba8f755a9e3eee8478ffa336091
--- /dev/null
+++ b/checkpoint-2000/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:10bc2585985998753712abc2ecadfff34b6e67963041968dba2cbe27e7cf0c3b
+size 14645
diff --git a/checkpoint-2000/scheduler.pt b/checkpoint-2000/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..10544ba513751c21a3b7a4f677046a56f6928804
--- /dev/null
+++ b/checkpoint-2000/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:26794880ede29b4548ba836f84d0aff6b385a2d83360206591e0210dd078bfba
+size 1465
diff --git a/checkpoint-2000/tokenizer.json b/checkpoint-2000/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-2000/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-2000/tokenizer_config.json b/checkpoint-2000/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-2000/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-2000/trainer_state.json b/checkpoint-2000/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..810731163f73c20b5478109528ac0b0ebcaf0f96
--- /dev/null
+++ b/checkpoint-2000/trainer_state.json
@@ -0,0 +1,734 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.5362469417166605,
+ "eval_steps": 500,
+ "global_step": 2000,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 2.458008027246551e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-2000/training_args.bin b/checkpoint-2000/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-2000/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-2200/README.md b/checkpoint-2200/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-2200/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-2200/adapter_config.json b/checkpoint-2200/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-2200/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-2200/adapter_model.safetensors b/checkpoint-2200/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..ddcf27fa4391a77e069c9bc06a898371057584e5
--- /dev/null
+++ b/checkpoint-2200/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c09137fa3c09d45e0ddaf3d37db926e20af9dc0d2d013715031618fa885a0148
+size 50360752
diff --git a/checkpoint-2200/chat_template.jinja b/checkpoint-2200/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-2200/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-2200/optimizer.pt b/checkpoint-2200/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..091e3e241e2ac36cbd21647ba4849703f1f4e098
--- /dev/null
+++ b/checkpoint-2200/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:7e34a7eafa5ffe9d595da548631104b7569a3c69d14d9203e2b0b697d0a2c028
+size 100828235
diff --git a/checkpoint-2200/rng_state.pth b/checkpoint-2200/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..203f470eceea2585413ceb87af9bf57b36dfb364
--- /dev/null
+++ b/checkpoint-2200/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5ae3ab067567845cbba15e0db0b98e81590592ebcaf63ef8185684a4e0f8715e
+size 14645
diff --git a/checkpoint-2200/scheduler.pt b/checkpoint-2200/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..e0c42fc5ac5eba1753ddf1ee9351c666ffcda21c
--- /dev/null
+++ b/checkpoint-2200/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3093f2c7ea7a67f370a885013a373c42b22b12bd84fae95a0221e821ecf823e9
+size 1465
diff --git a/checkpoint-2200/tokenizer.json b/checkpoint-2200/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-2200/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-2200/tokenizer_config.json b/checkpoint-2200/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-2200/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-2200/trainer_state.json b/checkpoint-2200/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..f2a91db473979bf0e8aa4adc01ccb99999aa30d0
--- /dev/null
+++ b/checkpoint-2200/trainer_state.json
@@ -0,0 +1,804 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.5898716358883266,
+ "eval_steps": 500,
+ "global_step": 2200,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 2.7065266797647462e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-2200/training_args.bin b/checkpoint-2200/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-2200/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-2400/README.md b/checkpoint-2400/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-2400/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-2400/adapter_config.json b/checkpoint-2400/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-2400/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-2400/adapter_model.safetensors b/checkpoint-2400/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..63055c8df26b5ba57706bb0c745ca72c62244ff5
--- /dev/null
+++ b/checkpoint-2400/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:352715f0b39309e4ee6508d2f855d41161303138355eb7548bcc6f459c746896
+size 50360752
diff --git a/checkpoint-2400/chat_template.jinja b/checkpoint-2400/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-2400/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-2400/optimizer.pt b/checkpoint-2400/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..7e0b798258393824e04b3b346f495b235fc1eb7a
--- /dev/null
+++ b/checkpoint-2400/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3e620d8259ac31856aa62169803645e6d592d7aee6db3b791a105ea999292d6d
+size 100828235
diff --git a/checkpoint-2400/rng_state.pth b/checkpoint-2400/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..c7857977dbc9a48f9ab58fb852430c8256622ec8
--- /dev/null
+++ b/checkpoint-2400/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:000f6ba765c74de08e259ce69b3eac86d97620f916260059014e36f6846c8df6
+size 14645
diff --git a/checkpoint-2400/scheduler.pt b/checkpoint-2400/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..82ca3ab98e9662810a5e482f3de93f87b42ca4fa
--- /dev/null
+++ b/checkpoint-2400/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1cc087535592c8cec31c3995ac9a31c01df67976c2c9eeb48eaf04a3830becc3
+size 1465
diff --git a/checkpoint-2400/tokenizer.json b/checkpoint-2400/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-2400/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-2400/tokenizer_config.json b/checkpoint-2400/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-2400/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-2400/trainer_state.json b/checkpoint-2400/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..c66f090ee621b717e737170dc06ce8549995bc70
--- /dev/null
+++ b/checkpoint-2400/trainer_state.json
@@ -0,0 +1,874 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.6434963300599926,
+ "eval_steps": 500,
+ "global_step": 2400,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 2.9507768981794406e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-2400/training_args.bin b/checkpoint-2400/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-2400/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-2600/README.md b/checkpoint-2600/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-2600/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-2600/adapter_config.json b/checkpoint-2600/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-2600/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-2600/adapter_model.safetensors b/checkpoint-2600/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..7afa743a254be266aca9413216656bfa7383f787
--- /dev/null
+++ b/checkpoint-2600/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d06bf39f110778bcb964a43bc1ef290d38366d43018a89d09d3a37e7e696cf57
+size 50360752
diff --git a/checkpoint-2600/chat_template.jinja b/checkpoint-2600/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-2600/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-2600/optimizer.pt b/checkpoint-2600/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..7514cdb8c6a1ee0d117963db6e89e63e6aeb00c6
--- /dev/null
+++ b/checkpoint-2600/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c4cd29ec1fcdb0431247aac514fab75ae6888193dc24cd42bb6920b78d722c8d
+size 100828235
diff --git a/checkpoint-2600/rng_state.pth b/checkpoint-2600/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..32083d5f553aa9ce09923cb77dc9e880bc79adfb
--- /dev/null
+++ b/checkpoint-2600/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:89b752ff6305664845d0a8f87ae70ed79c3c563144db28900d8a172b9dfc9780
+size 14645
diff --git a/checkpoint-2600/scheduler.pt b/checkpoint-2600/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..3b492052cdb19375a1d17b7622090617edfdedae
--- /dev/null
+++ b/checkpoint-2600/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f3606bff6312b72b3b51fbabcad96e085e08d05ec7ead01d6ccd499521240e71
+size 1465
diff --git a/checkpoint-2600/tokenizer.json b/checkpoint-2600/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-2600/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-2600/tokenizer_config.json b/checkpoint-2600/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-2600/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-2600/trainer_state.json b/checkpoint-2600/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..5684335c2435566712ce181a5c32cef8624a5bf9
--- /dev/null
+++ b/checkpoint-2600/trainer_state.json
@@ -0,0 +1,944 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.6971210242316587,
+ "eval_steps": 500,
+ "global_step": 2600,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 3.197326861503836e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-2600/training_args.bin b/checkpoint-2600/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-2600/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-2800/README.md b/checkpoint-2800/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-2800/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-2800/adapter_config.json b/checkpoint-2800/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-2800/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-2800/adapter_model.safetensors b/checkpoint-2800/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..4237a2b77542cf2b1905c960c34bceae1ef6a0e4
--- /dev/null
+++ b/checkpoint-2800/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:9341c2981beb59a12df8327cd90b3489aa5a1c57abdae2072b03804ab838b5e6
+size 50360752
diff --git a/checkpoint-2800/chat_template.jinja b/checkpoint-2800/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-2800/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-2800/optimizer.pt b/checkpoint-2800/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..dbf5f4fec877e7d07c12999b88b1e371b9d44d78
--- /dev/null
+++ b/checkpoint-2800/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:02fc4d956c8fbe3378240b6790ff74c8e0253a10478d1fbaeba92d4ec870f7fd
+size 100828235
diff --git a/checkpoint-2800/rng_state.pth b/checkpoint-2800/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..f68aafde8aac62ace2f4d7890d7bdfb76782b4e0
--- /dev/null
+++ b/checkpoint-2800/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d5f657d76b555833134e4fa8034da77eac4639aecb8f189e14cd3669258bc7c8
+size 14645
diff --git a/checkpoint-2800/scheduler.pt b/checkpoint-2800/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..a1cf24b8870a2ddd09c47e9e33d1efa1fe4a314e
--- /dev/null
+++ b/checkpoint-2800/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d2fd4f05577235783386ce72862134bcceb579f6782c2b37884bd0c38983b508
+size 1465
diff --git a/checkpoint-2800/tokenizer.json b/checkpoint-2800/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-2800/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-2800/tokenizer_config.json b/checkpoint-2800/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-2800/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-2800/trainer_state.json b/checkpoint-2800/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..70bed6ffb1cbef27a91668c4a5bbc136cb5eb658
--- /dev/null
+++ b/checkpoint-2800/trainer_state.json
@@ -0,0 +1,1014 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.7507457184033247,
+ "eval_steps": 500,
+ "global_step": 2800,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 3.4429710429456384e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-2800/training_args.bin b/checkpoint-2800/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-2800/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-3000/README.md b/checkpoint-3000/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-3000/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-3000/adapter_config.json b/checkpoint-3000/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-3000/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-3000/adapter_model.safetensors b/checkpoint-3000/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..dc6e448b857b020f1bed075d512499b096ec0243
--- /dev/null
+++ b/checkpoint-3000/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:502d5f8ea7c93de3267c6c653b1f66decc4973a650ea547afb6815aa59e41c00
+size 50360752
diff --git a/checkpoint-3000/chat_template.jinja b/checkpoint-3000/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-3000/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-3000/optimizer.pt b/checkpoint-3000/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..a1b96a854f7716fe38d440feb2c6a459a76a15be
--- /dev/null
+++ b/checkpoint-3000/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1699182c47e8d73978a11f7054ce75e1f2000cd96529a49467acad54b95ad137
+size 100828235
diff --git a/checkpoint-3000/rng_state.pth b/checkpoint-3000/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..ffd5db9870a47a2e580551277cf767721b3c7a60
--- /dev/null
+++ b/checkpoint-3000/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5d51c550d5e4ed5ea74f62db61d558c3947bf0e7be99d1f8aa1641764a2899d9
+size 14645
diff --git a/checkpoint-3000/scheduler.pt b/checkpoint-3000/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..acac253e62277486eb722d8fa0fe59424e49a316
--- /dev/null
+++ b/checkpoint-3000/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c82aebb2adaf30bcd259ae87af34af15cdbc12ff481b004f2a325924ae650bc2
+size 1465
diff --git a/checkpoint-3000/tokenizer.json b/checkpoint-3000/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-3000/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-3000/tokenizer_config.json b/checkpoint-3000/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-3000/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-3000/trainer_state.json b/checkpoint-3000/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..1c2ba4aedfcce97b7d14ed61549e18d461fef833
--- /dev/null
+++ b/checkpoint-3000/trainer_state.json
@@ -0,0 +1,1084 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.8043704125749908,
+ "eval_steps": 500,
+ "global_step": 3000,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 3.6907443999815885e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-3000/training_args.bin b/checkpoint-3000/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-3000/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-3200/README.md b/checkpoint-3200/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-3200/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-3200/adapter_config.json b/checkpoint-3200/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-3200/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-3200/adapter_model.safetensors b/checkpoint-3200/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..a69e00ef65f9370d127c731e96e1f03c1be3d511
--- /dev/null
+++ b/checkpoint-3200/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:67aab25c78f3ff33249c13ac8ab9e17170269d9733170476f5b00449192d4df5
+size 50360752
diff --git a/checkpoint-3200/chat_template.jinja b/checkpoint-3200/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-3200/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-3200/optimizer.pt b/checkpoint-3200/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..885fbbfb73db64400cc5439d50df267b61573099
--- /dev/null
+++ b/checkpoint-3200/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:aaec4ae388a2a8f201cb5850524c13702b2f57da221015fcf162dc54ffb5db5f
+size 100828235
diff --git a/checkpoint-3200/rng_state.pth b/checkpoint-3200/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..7cecc0ab50e8cec158fa68ca2266b4fe3ff95538
--- /dev/null
+++ b/checkpoint-3200/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8329ae73ab98313c3f32048b8136bc9c1d53fe02a9d15fe134568aff617eb2ab
+size 14645
diff --git a/checkpoint-3200/scheduler.pt b/checkpoint-3200/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..334db1664d5ec95c81e46299d343bad4d6056ee7
--- /dev/null
+++ b/checkpoint-3200/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:03c0e2a76cba07205e0fa06795ebf8be7d64722b16bcfeacafeb1fdc874a2023
+size 1465
diff --git a/checkpoint-3200/tokenizer.json b/checkpoint-3200/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-3200/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-3200/tokenizer_config.json b/checkpoint-3200/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-3200/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-3200/trainer_state.json b/checkpoint-3200/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..a9708a9c6d565378ba5892c80d8a2896595861c5
--- /dev/null
+++ b/checkpoint-3200/trainer_state.json
@@ -0,0 +1,1154 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.8579951067466568,
+ "eval_steps": 500,
+ "global_step": 3200,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 3.938011930771415e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-3200/training_args.bin b/checkpoint-3200/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-3200/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-3400/README.md b/checkpoint-3400/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-3400/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-3400/adapter_config.json b/checkpoint-3400/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-3400/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-3400/adapter_model.safetensors b/checkpoint-3400/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..a47e91eecdc4a973d0b317567c9f32d3dc258ac9
--- /dev/null
+++ b/checkpoint-3400/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c3ec0fd4398b542e929710aa03ee5201c0a58bb48c61514fbe9f243651bb66eb
+size 50360752
diff --git a/checkpoint-3400/chat_template.jinja b/checkpoint-3400/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-3400/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-3400/optimizer.pt b/checkpoint-3400/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..f683ff93081c04db5cd29dbd813394ee1eeb3084
--- /dev/null
+++ b/checkpoint-3400/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:67e45e698965aef6a51a58f0bb82a862bcc1242c2b123ba7238328be4399f0f1
+size 100828235
diff --git a/checkpoint-3400/rng_state.pth b/checkpoint-3400/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..cddc21f632fdbacd9e9bc19e13ce72e2a3d1ac75
--- /dev/null
+++ b/checkpoint-3400/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:09b46f94b951b86129463f17a330ae11cf0c1eae5abf93203d5b896e6e628d4f
+size 14645
diff --git a/checkpoint-3400/scheduler.pt b/checkpoint-3400/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..63c37ca3ad171bf5a6ad23c0273dee7aac46918c
--- /dev/null
+++ b/checkpoint-3400/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a290cdb73f982dc9af70b364cd219af8450564e4a10c0cd208c0c9d4bd2215c0
+size 1465
diff --git a/checkpoint-3400/tokenizer.json b/checkpoint-3400/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-3400/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-3400/tokenizer_config.json b/checkpoint-3400/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-3400/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-3400/trainer_state.json b/checkpoint-3400/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..3aa05e25dc1f3b976cca01136110820fa99b7b4e
--- /dev/null
+++ b/checkpoint-3400/trainer_state.json
@@ -0,0 +1,1224 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.9116198009183228,
+ "eval_steps": 500,
+ "global_step": 3400,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 4.185686979384115e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-3400/training_args.bin b/checkpoint-3400/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-3400/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-3600/README.md b/checkpoint-3600/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-3600/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-3600/adapter_config.json b/checkpoint-3600/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-3600/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-3600/adapter_model.safetensors b/checkpoint-3600/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..6398b3c31d809f747a915a887f8f582249aa209c
--- /dev/null
+++ b/checkpoint-3600/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:fee7278ec9639f4bb3f7192769eadf94307962b5bfa164f7ce0398f2f7b1fbdc
+size 50360752
diff --git a/checkpoint-3600/chat_template.jinja b/checkpoint-3600/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-3600/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-3600/optimizer.pt b/checkpoint-3600/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..cf6731ef0c0288544ead46febe9f6ae45ccdb73f
--- /dev/null
+++ b/checkpoint-3600/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:40c884e18d08a65d9f227d635573d0bae9ddceb241bd4dcb037e094c42034863
+size 100828235
diff --git a/checkpoint-3600/rng_state.pth b/checkpoint-3600/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..71b7ff094fad921a4ad832309245545666aca960
--- /dev/null
+++ b/checkpoint-3600/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a8829d93dfacc727ad08419cac075580443395727c7aeb155053bf5133bb6ead
+size 14645
diff --git a/checkpoint-3600/scheduler.pt b/checkpoint-3600/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..e636f3df6836282e54c8f286e5395906d5ae6b6e
--- /dev/null
+++ b/checkpoint-3600/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:61368de55afbac7eb28a60f8c94942258b7873b811ea6e873d61969c349866d0
+size 1465
diff --git a/checkpoint-3600/tokenizer.json b/checkpoint-3600/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-3600/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-3600/tokenizer_config.json b/checkpoint-3600/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-3600/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-3600/trainer_state.json b/checkpoint-3600/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..0d75dc2ac528f07d8ca5a3121f82208d79bad378
--- /dev/null
+++ b/checkpoint-3600/trainer_state.json
@@ -0,0 +1,1294 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.9652444950899889,
+ "eval_steps": 500,
+ "global_step": 3600,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 4.4337636641191526e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-3600/training_args.bin b/checkpoint-3600/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-3600/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-3800/README.md b/checkpoint-3800/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-3800/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-3800/adapter_config.json b/checkpoint-3800/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-3800/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-3800/adapter_model.safetensors b/checkpoint-3800/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..af78446644bf6ef597d8d4875a8d9ddda753f59c
--- /dev/null
+++ b/checkpoint-3800/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a47dc8a237bb61b99d278738597538aa15c7ddadf8e039f10ae9b5e8b368e3b5
+size 50360752
diff --git a/checkpoint-3800/chat_template.jinja b/checkpoint-3800/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-3800/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-3800/optimizer.pt b/checkpoint-3800/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..d12ff8aaa7a84f7433a99f0eea7d0c332c2e1f61
--- /dev/null
+++ b/checkpoint-3800/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:7c32f7e4ae37b7239e5cb7e2a846d23dd7a5456d971b1f112afc6bb6c33382c0
+size 100828235
diff --git a/checkpoint-3800/rng_state.pth b/checkpoint-3800/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..4d9a3c4151e728d94a04a4bf0f74933e0b40801b
--- /dev/null
+++ b/checkpoint-3800/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8035a28d94e2861444b60e73202ade9ad8d6b5e47e931e266011fcb185ffccda
+size 14645
diff --git a/checkpoint-3800/scheduler.pt b/checkpoint-3800/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..05ac5dcdb42f40390901372bbc1024100a539c6c
--- /dev/null
+++ b/checkpoint-3800/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d0d71bc291351c30a9b43de3f877c602b9454601b2d8ae2d40b580fad0c372ec
+size 1465
diff --git a/checkpoint-3800/tokenizer.json b/checkpoint-3800/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-3800/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-3800/tokenizer_config.json b/checkpoint-3800/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-3800/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-3800/trainer_state.json b/checkpoint-3800/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..352c401fd44dd4ca31d0a528c512cb989b83c6d0
--- /dev/null
+++ b/checkpoint-3800/trainer_state.json
@@ -0,0 +1,1364 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.018768642960083,
+ "eval_steps": 500,
+ "global_step": 3800,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 4.6827452904938496e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-3800/training_args.bin b/checkpoint-3800/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-3800/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-400/README.md b/checkpoint-400/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-400/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-400/adapter_config.json b/checkpoint-400/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-400/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-400/adapter_model.safetensors b/checkpoint-400/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..41ab37938d8249ea07308ff318df96b491e75b4c
--- /dev/null
+++ b/checkpoint-400/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:dc1f8c7a93a803cf603c10f53f449d2defafb9920543481c424e82b6afb4c7d6
+size 50360752
diff --git a/checkpoint-400/chat_template.jinja b/checkpoint-400/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-400/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-400/optimizer.pt b/checkpoint-400/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..8d96db457b0917b58faa719dae0868ad7e67f55c
--- /dev/null
+++ b/checkpoint-400/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1a793075733206458d44e62e354e6175a015e55dfa0f8ac76da92f1c47e2dbc0
+size 100828235
diff --git a/checkpoint-400/rng_state.pth b/checkpoint-400/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..8fd56cdfd0ceef28c611428fff47f7e2e6a93ac6
--- /dev/null
+++ b/checkpoint-400/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d4d66c1b8ffb7bff6a5dbad802a36584713b4acb078c71e97b0d664ee0cea1ec
+size 14645
diff --git a/checkpoint-400/scheduler.pt b/checkpoint-400/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..f69bd703793daca9575464b2aef6843304855cfd
--- /dev/null
+++ b/checkpoint-400/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5105cfc7d8f214e6052469b04036343fd59f9ed025860545346c8ee4bab7ffbb
+size 1465
diff --git a/checkpoint-400/tokenizer.json b/checkpoint-400/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-400/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-400/tokenizer_config.json b/checkpoint-400/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-400/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-400/trainer_state.json b/checkpoint-400/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..cc0ff0dd56703e9b4783585fdb2fb08369efe9bb
--- /dev/null
+++ b/checkpoint-400/trainer_state.json
@@ -0,0 +1,174 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.1072493883433321,
+ "eval_steps": 500,
+ "global_step": 400,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 4.898960803423642e+16,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-400/training_args.bin b/checkpoint-400/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-400/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-4000/README.md b/checkpoint-4000/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-4000/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-4000/adapter_config.json b/checkpoint-4000/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-4000/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-4000/adapter_model.safetensors b/checkpoint-4000/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..c2f292866793d527ea4da26c15700a0567a0b036
--- /dev/null
+++ b/checkpoint-4000/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1753129c326465b61b9f2b91512603e563347c696c7470840e9383d276ed1354
+size 50360752
diff --git a/checkpoint-4000/chat_template.jinja b/checkpoint-4000/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-4000/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-4000/optimizer.pt b/checkpoint-4000/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..40114c754025380955dec8320709ae225624f2c2
--- /dev/null
+++ b/checkpoint-4000/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ba9d6810641b1594a7f3512ff5b36560c25767c454d69e27ad89d0424ffe09f4
+size 100828235
diff --git a/checkpoint-4000/rng_state.pth b/checkpoint-4000/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..541abbdf017f3e2224388fb36a6a5483c3b53659
--- /dev/null
+++ b/checkpoint-4000/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e2c135a732e64620e2c2d8601198a5da6963f58eea6921151b2d3a78fc57af9c
+size 14645
diff --git a/checkpoint-4000/scheduler.pt b/checkpoint-4000/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..e2db81d47a471a7184d394eb56ed90d0f9fc1930
--- /dev/null
+++ b/checkpoint-4000/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:675f310dd1e4b16f9399e9e931f24613c0f68080a472c0dc6e53f616e9348975
+size 1465
diff --git a/checkpoint-4000/tokenizer.json b/checkpoint-4000/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-4000/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-4000/tokenizer_config.json b/checkpoint-4000/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-4000/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-4000/trainer_state.json b/checkpoint-4000/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..550195b1256ad51dd413a0a7b14045aaa85c6648
--- /dev/null
+++ b/checkpoint-4000/trainer_state.json
@@ -0,0 +1,1434 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.0723933371317491,
+ "eval_steps": 500,
+ "global_step": 4000,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 4.92759796309205e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-4000/training_args.bin b/checkpoint-4000/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-4000/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-4200/README.md b/checkpoint-4200/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-4200/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-4200/adapter_config.json b/checkpoint-4200/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-4200/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-4200/adapter_model.safetensors b/checkpoint-4200/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..9c2876da9e4a78b08a1b3fb9c72fc69dc0ffe627
--- /dev/null
+++ b/checkpoint-4200/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1218b5a13ffd881be6928bc76e8e1f73bea0e216013407f9f975c7b4070cdc95
+size 50360752
diff --git a/checkpoint-4200/chat_template.jinja b/checkpoint-4200/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-4200/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-4200/optimizer.pt b/checkpoint-4200/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..f4e46f559904761f3b489621ffcc439aca1cb140
--- /dev/null
+++ b/checkpoint-4200/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b08a9710b97c94c55a68cb4cc4cecb3cb94b0ac69d2b39f2bb0a34c3a6af83d3
+size 100828235
diff --git a/checkpoint-4200/rng_state.pth b/checkpoint-4200/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..fe6f5e3d5dafb37629d3b895c26352eaa5e69572
--- /dev/null
+++ b/checkpoint-4200/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e8cd3600f35fd154e02403d9b006d3301abebfea18794563b39b24b771c5cb36
+size 14645
diff --git a/checkpoint-4200/scheduler.pt b/checkpoint-4200/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..5cb5adc051524746891ca5dbcc76bd65e7da7a83
--- /dev/null
+++ b/checkpoint-4200/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b719e1a8e991070bbe2aaa2873e36bc1c032a531a35c0a22d83ba6170bea31ef
+size 1465
diff --git a/checkpoint-4200/tokenizer.json b/checkpoint-4200/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-4200/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-4200/tokenizer_config.json b/checkpoint-4200/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-4200/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-4200/trainer_state.json b/checkpoint-4200/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..f250b790a39dcd909519a3371f1a95c7cfe24baa
--- /dev/null
+++ b/checkpoint-4200/trainer_state.json
@@ -0,0 +1,1504 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.1260180313034152,
+ "eval_steps": 500,
+ "global_step": 4200,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 5.176267859338322e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-4200/training_args.bin b/checkpoint-4200/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-4200/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-4400/README.md b/checkpoint-4400/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-4400/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-4400/adapter_config.json b/checkpoint-4400/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-4400/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-4400/adapter_model.safetensors b/checkpoint-4400/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..d184d7f7c1dec481dc035df11399a7c7cf864489
--- /dev/null
+++ b/checkpoint-4400/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5825dc7f3591648e0cf7c4ff0842260cc574133c333db11e4899294019d8e09e
+size 50360752
diff --git a/checkpoint-4400/chat_template.jinja b/checkpoint-4400/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-4400/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-4400/optimizer.pt b/checkpoint-4400/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..2e51386cca4a04bc34f24e31d2dbfcd44b5379e5
--- /dev/null
+++ b/checkpoint-4400/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:2756659988e3b82b58e6cfeac4948da01697beee526da77cc0e8db9d1f1cc18d
+size 100828235
diff --git a/checkpoint-4400/rng_state.pth b/checkpoint-4400/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..65a800071b3540e6d69cd8b21d9f47d109be820f
--- /dev/null
+++ b/checkpoint-4400/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:15298ff903da8de8196d202d25f163414c9c0666065ac1b553486a4e00a5c028
+size 14645
diff --git a/checkpoint-4400/scheduler.pt b/checkpoint-4400/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..9cd48156ef9bdbfd2f9abb3f2f4c167d88a561ae
--- /dev/null
+++ b/checkpoint-4400/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:12b034b4b0107f8077e82d6949164cc02311d2cc9a4806d94aee238c7cb72f6e
+size 1465
diff --git a/checkpoint-4400/tokenizer.json b/checkpoint-4400/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-4400/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-4400/tokenizer_config.json b/checkpoint-4400/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-4400/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-4400/trainer_state.json b/checkpoint-4400/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..c29080eddb98d20bc4e59127ddce73f237d281fe
--- /dev/null
+++ b/checkpoint-4400/trainer_state.json
@@ -0,0 +1,1574 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.1796427254750812,
+ "eval_steps": 500,
+ "global_step": 4400,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 5.42165408619946e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-4400/training_args.bin b/checkpoint-4400/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-4400/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-4600/README.md b/checkpoint-4600/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-4600/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-4600/adapter_config.json b/checkpoint-4600/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-4600/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-4600/adapter_model.safetensors b/checkpoint-4600/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..4988d25e0054a39486e5f6c97dfe1342d3d8317e
--- /dev/null
+++ b/checkpoint-4600/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21717bae5311060418f1c58765d4605b8c8ab67a5c650469322feff8f436eddc
+size 50360752
diff --git a/checkpoint-4600/chat_template.jinja b/checkpoint-4600/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-4600/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-4600/optimizer.pt b/checkpoint-4600/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..efe67fdf9fd13acfaad8ddd1a472042bbaa6b6a3
--- /dev/null
+++ b/checkpoint-4600/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:01a56e63c21a13dd2832bdd26f3995ece9939bd91ab4bd5966fd458a6834f1a5
+size 100828235
diff --git a/checkpoint-4600/rng_state.pth b/checkpoint-4600/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..7385973febfa7df5313c696b7dbf0b9e4bd8ba8f
--- /dev/null
+++ b/checkpoint-4600/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:93f56a983a825a4d012386c1eb49c8b7340c83bb0e7b0250c508ac173d9c8d1e
+size 14645
diff --git a/checkpoint-4600/scheduler.pt b/checkpoint-4600/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..ecf27ac3eaf20a282860dd567a6dc7658610106b
--- /dev/null
+++ b/checkpoint-4600/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:2e54391a300f76ab81ca47c9b5dc0fa1d5e3a764ba448cc3994ba625dd036edc
+size 1465
diff --git a/checkpoint-4600/tokenizer.json b/checkpoint-4600/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-4600/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-4600/tokenizer_config.json b/checkpoint-4600/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-4600/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-4600/trainer_state.json b/checkpoint-4600/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..3a2759a7c48e2532e2de5623ea1dbbd6bbabf9db
--- /dev/null
+++ b/checkpoint-4600/trainer_state.json
@@ -0,0 +1,1644 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.2332674196467472,
+ "eval_steps": 500,
+ "global_step": 4600,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ },
+ {
+ "epoch": 1.185005194892248,
+ "grad_norm": 0.16667844355106354,
+ "learning_rate": 6.114611260053619e-06,
+ "loss": 0.5159809589385986,
+ "step": 4420
+ },
+ {
+ "epoch": 1.1903676643094145,
+ "grad_norm": 0.19206570088863373,
+ "learning_rate": 6.074396782841823e-06,
+ "loss": 0.47541089057922364,
+ "step": 4440
+ },
+ {
+ "epoch": 1.195730133726581,
+ "grad_norm": 0.17394617199897766,
+ "learning_rate": 6.034182305630027e-06,
+ "loss": 0.5470661640167236,
+ "step": 4460
+ },
+ {
+ "epoch": 1.2010926031437477,
+ "grad_norm": 0.210404634475708,
+ "learning_rate": 5.993967828418231e-06,
+ "loss": 0.5377882957458496,
+ "step": 4480
+ },
+ {
+ "epoch": 1.2064550725609142,
+ "grad_norm": 0.18084648251533508,
+ "learning_rate": 5.953753351206435e-06,
+ "loss": 0.5037185192108155,
+ "step": 4500
+ },
+ {
+ "epoch": 1.211817541978081,
+ "grad_norm": 0.23707027733325958,
+ "learning_rate": 5.913538873994639e-06,
+ "loss": 0.4822190284729004,
+ "step": 4520
+ },
+ {
+ "epoch": 1.2171800113952476,
+ "grad_norm": 0.16474473476409912,
+ "learning_rate": 5.873324396782842e-06,
+ "loss": 0.46645288467407225,
+ "step": 4540
+ },
+ {
+ "epoch": 1.2225424808124141,
+ "grad_norm": 0.2142348438501358,
+ "learning_rate": 5.833109919571046e-06,
+ "loss": 0.5255855560302735,
+ "step": 4560
+ },
+ {
+ "epoch": 1.2279049502295807,
+ "grad_norm": 0.2531765103340149,
+ "learning_rate": 5.79289544235925e-06,
+ "loss": 0.507044792175293,
+ "step": 4580
+ },
+ {
+ "epoch": 1.2332674196467472,
+ "grad_norm": 0.2553550899028778,
+ "learning_rate": 5.7526809651474535e-06,
+ "loss": 0.4767824649810791,
+ "step": 4600
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 5.6720162317143245e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-4600/training_args.bin b/checkpoint-4600/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-4600/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-4800/README.md b/checkpoint-4800/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-4800/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-4800/adapter_config.json b/checkpoint-4800/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-4800/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-4800/adapter_model.safetensors b/checkpoint-4800/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..4c6b8ee9313570c4d510898f3e7ec7e7df2572e2
--- /dev/null
+++ b/checkpoint-4800/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d9513a26ca3a56a159421394d3c0b2f38b7abbaf02ac994237dad40f0771795e
+size 50360752
diff --git a/checkpoint-4800/chat_template.jinja b/checkpoint-4800/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-4800/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-4800/optimizer.pt b/checkpoint-4800/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..eab264233d70f1f033b283ba6d37afb312b04219
--- /dev/null
+++ b/checkpoint-4800/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:960b7f6c6cd995ac69141e48c170ed41272c7a3dce50a75ea8ec64d985e90bb9
+size 100828235
diff --git a/checkpoint-4800/rng_state.pth b/checkpoint-4800/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..26680a6db664a93f70bb4f430d5c58e8e318b85c
--- /dev/null
+++ b/checkpoint-4800/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0d8812583303a1ffd00675d7bed0ba594fe1235211733de354da4adbe001aa26
+size 14645
diff --git a/checkpoint-4800/scheduler.pt b/checkpoint-4800/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..221678c25457523727c5029ee0b04a7b46300d52
--- /dev/null
+++ b/checkpoint-4800/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f05c4d2a4a26723c2086c10e751650c4a1e3b250d170a087405862de4b2116f1
+size 1465
diff --git a/checkpoint-4800/tokenizer.json b/checkpoint-4800/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-4800/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-4800/tokenizer_config.json b/checkpoint-4800/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-4800/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-4800/trainer_state.json b/checkpoint-4800/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..01c6cf43b446916300a5ef3c554419ec8700794e
--- /dev/null
+++ b/checkpoint-4800/trainer_state.json
@@ -0,0 +1,1714 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.2868921138184133,
+ "eval_steps": 500,
+ "global_step": 4800,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ },
+ {
+ "epoch": 1.185005194892248,
+ "grad_norm": 0.16667844355106354,
+ "learning_rate": 6.114611260053619e-06,
+ "loss": 0.5159809589385986,
+ "step": 4420
+ },
+ {
+ "epoch": 1.1903676643094145,
+ "grad_norm": 0.19206570088863373,
+ "learning_rate": 6.074396782841823e-06,
+ "loss": 0.47541089057922364,
+ "step": 4440
+ },
+ {
+ "epoch": 1.195730133726581,
+ "grad_norm": 0.17394617199897766,
+ "learning_rate": 6.034182305630027e-06,
+ "loss": 0.5470661640167236,
+ "step": 4460
+ },
+ {
+ "epoch": 1.2010926031437477,
+ "grad_norm": 0.210404634475708,
+ "learning_rate": 5.993967828418231e-06,
+ "loss": 0.5377882957458496,
+ "step": 4480
+ },
+ {
+ "epoch": 1.2064550725609142,
+ "grad_norm": 0.18084648251533508,
+ "learning_rate": 5.953753351206435e-06,
+ "loss": 0.5037185192108155,
+ "step": 4500
+ },
+ {
+ "epoch": 1.211817541978081,
+ "grad_norm": 0.23707027733325958,
+ "learning_rate": 5.913538873994639e-06,
+ "loss": 0.4822190284729004,
+ "step": 4520
+ },
+ {
+ "epoch": 1.2171800113952476,
+ "grad_norm": 0.16474473476409912,
+ "learning_rate": 5.873324396782842e-06,
+ "loss": 0.46645288467407225,
+ "step": 4540
+ },
+ {
+ "epoch": 1.2225424808124141,
+ "grad_norm": 0.2142348438501358,
+ "learning_rate": 5.833109919571046e-06,
+ "loss": 0.5255855560302735,
+ "step": 4560
+ },
+ {
+ "epoch": 1.2279049502295807,
+ "grad_norm": 0.2531765103340149,
+ "learning_rate": 5.79289544235925e-06,
+ "loss": 0.507044792175293,
+ "step": 4580
+ },
+ {
+ "epoch": 1.2332674196467472,
+ "grad_norm": 0.2553550899028778,
+ "learning_rate": 5.7526809651474535e-06,
+ "loss": 0.4767824649810791,
+ "step": 4600
+ },
+ {
+ "epoch": 1.238629889063914,
+ "grad_norm": 0.14484412968158722,
+ "learning_rate": 5.712466487935657e-06,
+ "loss": 0.4675601005554199,
+ "step": 4620
+ },
+ {
+ "epoch": 1.2439923584810806,
+ "grad_norm": 0.14328251779079437,
+ "learning_rate": 5.672252010723861e-06,
+ "loss": 0.4956005573272705,
+ "step": 4640
+ },
+ {
+ "epoch": 1.2493548278982471,
+ "grad_norm": 0.1739245355129242,
+ "learning_rate": 5.632037533512065e-06,
+ "loss": 0.48583345413208007,
+ "step": 4660
+ },
+ {
+ "epoch": 1.2547172973154137,
+ "grad_norm": 0.21294184029102325,
+ "learning_rate": 5.591823056300268e-06,
+ "loss": 0.520921277999878,
+ "step": 4680
+ },
+ {
+ "epoch": 1.2600797667325803,
+ "grad_norm": 0.25132355093955994,
+ "learning_rate": 5.551608579088472e-06,
+ "loss": 0.5295385837554931,
+ "step": 4700
+ },
+ {
+ "epoch": 1.265442236149747,
+ "grad_norm": 0.18603841960430145,
+ "learning_rate": 5.511394101876676e-06,
+ "loss": 0.47570199966430665,
+ "step": 4720
+ },
+ {
+ "epoch": 1.2708047055669136,
+ "grad_norm": 0.19883134961128235,
+ "learning_rate": 5.471179624664879e-06,
+ "loss": 0.5016080379486084,
+ "step": 4740
+ },
+ {
+ "epoch": 1.2761671749840802,
+ "grad_norm": 0.19640181958675385,
+ "learning_rate": 5.430965147453083e-06,
+ "loss": 0.4999081134796143,
+ "step": 4760
+ },
+ {
+ "epoch": 1.2815296444012467,
+ "grad_norm": 0.2584764361381531,
+ "learning_rate": 5.390750670241287e-06,
+ "loss": 0.4780082702636719,
+ "step": 4780
+ },
+ {
+ "epoch": 1.2868921138184133,
+ "grad_norm": 0.2925741374492645,
+ "learning_rate": 5.3505361930294905e-06,
+ "loss": 0.5131395816802978,
+ "step": 4800
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 5.917521773072056e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-4800/training_args.bin b/checkpoint-4800/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-4800/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-5000/README.md b/checkpoint-5000/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-5000/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-5000/adapter_config.json b/checkpoint-5000/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-5000/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-5000/adapter_model.safetensors b/checkpoint-5000/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..034de777b048107239afb5af185aae9eecc735e2
--- /dev/null
+++ b/checkpoint-5000/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:03a2a5c7ca7066503fa16253b698224bb6f1203f6ca8894cd8fb343d11b788e3
+size 50360752
diff --git a/checkpoint-5000/chat_template.jinja b/checkpoint-5000/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-5000/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-5000/optimizer.pt b/checkpoint-5000/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..70817699ea05b83b2189e047b2cd061bd86f1850
--- /dev/null
+++ b/checkpoint-5000/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5334c60b6c07b131dd4b79a5db0b6852a3cdd7c9ca8a02d059967a8a4e2efdf9
+size 100828235
diff --git a/checkpoint-5000/rng_state.pth b/checkpoint-5000/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..6a292701c0e5e8c7bf505abb23c59ce1e2f3efbe
--- /dev/null
+++ b/checkpoint-5000/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:660f06b1a6ade2b46bec0a145e16ffee761d68f863ea390644961bd21c7e83dd
+size 14645
diff --git a/checkpoint-5000/scheduler.pt b/checkpoint-5000/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..9010b7e60015c12ec22b08e77c1ca8781502c2a9
--- /dev/null
+++ b/checkpoint-5000/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d75f431e14a5ca6bd149963d8053337b6fecbfb6654d3108b1accb7e1005dc39
+size 1465
diff --git a/checkpoint-5000/tokenizer.json b/checkpoint-5000/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-5000/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-5000/tokenizer_config.json b/checkpoint-5000/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-5000/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-5000/trainer_state.json b/checkpoint-5000/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..f78b35c5e703a5c9f4a3f9b87f42c57f37f6bfb4
--- /dev/null
+++ b/checkpoint-5000/trainer_state.json
@@ -0,0 +1,1784 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.3405168079900793,
+ "eval_steps": 500,
+ "global_step": 5000,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ },
+ {
+ "epoch": 1.185005194892248,
+ "grad_norm": 0.16667844355106354,
+ "learning_rate": 6.114611260053619e-06,
+ "loss": 0.5159809589385986,
+ "step": 4420
+ },
+ {
+ "epoch": 1.1903676643094145,
+ "grad_norm": 0.19206570088863373,
+ "learning_rate": 6.074396782841823e-06,
+ "loss": 0.47541089057922364,
+ "step": 4440
+ },
+ {
+ "epoch": 1.195730133726581,
+ "grad_norm": 0.17394617199897766,
+ "learning_rate": 6.034182305630027e-06,
+ "loss": 0.5470661640167236,
+ "step": 4460
+ },
+ {
+ "epoch": 1.2010926031437477,
+ "grad_norm": 0.210404634475708,
+ "learning_rate": 5.993967828418231e-06,
+ "loss": 0.5377882957458496,
+ "step": 4480
+ },
+ {
+ "epoch": 1.2064550725609142,
+ "grad_norm": 0.18084648251533508,
+ "learning_rate": 5.953753351206435e-06,
+ "loss": 0.5037185192108155,
+ "step": 4500
+ },
+ {
+ "epoch": 1.211817541978081,
+ "grad_norm": 0.23707027733325958,
+ "learning_rate": 5.913538873994639e-06,
+ "loss": 0.4822190284729004,
+ "step": 4520
+ },
+ {
+ "epoch": 1.2171800113952476,
+ "grad_norm": 0.16474473476409912,
+ "learning_rate": 5.873324396782842e-06,
+ "loss": 0.46645288467407225,
+ "step": 4540
+ },
+ {
+ "epoch": 1.2225424808124141,
+ "grad_norm": 0.2142348438501358,
+ "learning_rate": 5.833109919571046e-06,
+ "loss": 0.5255855560302735,
+ "step": 4560
+ },
+ {
+ "epoch": 1.2279049502295807,
+ "grad_norm": 0.2531765103340149,
+ "learning_rate": 5.79289544235925e-06,
+ "loss": 0.507044792175293,
+ "step": 4580
+ },
+ {
+ "epoch": 1.2332674196467472,
+ "grad_norm": 0.2553550899028778,
+ "learning_rate": 5.7526809651474535e-06,
+ "loss": 0.4767824649810791,
+ "step": 4600
+ },
+ {
+ "epoch": 1.238629889063914,
+ "grad_norm": 0.14484412968158722,
+ "learning_rate": 5.712466487935657e-06,
+ "loss": 0.4675601005554199,
+ "step": 4620
+ },
+ {
+ "epoch": 1.2439923584810806,
+ "grad_norm": 0.14328251779079437,
+ "learning_rate": 5.672252010723861e-06,
+ "loss": 0.4956005573272705,
+ "step": 4640
+ },
+ {
+ "epoch": 1.2493548278982471,
+ "grad_norm": 0.1739245355129242,
+ "learning_rate": 5.632037533512065e-06,
+ "loss": 0.48583345413208007,
+ "step": 4660
+ },
+ {
+ "epoch": 1.2547172973154137,
+ "grad_norm": 0.21294184029102325,
+ "learning_rate": 5.591823056300268e-06,
+ "loss": 0.520921277999878,
+ "step": 4680
+ },
+ {
+ "epoch": 1.2600797667325803,
+ "grad_norm": 0.25132355093955994,
+ "learning_rate": 5.551608579088472e-06,
+ "loss": 0.5295385837554931,
+ "step": 4700
+ },
+ {
+ "epoch": 1.265442236149747,
+ "grad_norm": 0.18603841960430145,
+ "learning_rate": 5.511394101876676e-06,
+ "loss": 0.47570199966430665,
+ "step": 4720
+ },
+ {
+ "epoch": 1.2708047055669136,
+ "grad_norm": 0.19883134961128235,
+ "learning_rate": 5.471179624664879e-06,
+ "loss": 0.5016080379486084,
+ "step": 4740
+ },
+ {
+ "epoch": 1.2761671749840802,
+ "grad_norm": 0.19640181958675385,
+ "learning_rate": 5.430965147453083e-06,
+ "loss": 0.4999081134796143,
+ "step": 4760
+ },
+ {
+ "epoch": 1.2815296444012467,
+ "grad_norm": 0.2584764361381531,
+ "learning_rate": 5.390750670241287e-06,
+ "loss": 0.4780082702636719,
+ "step": 4780
+ },
+ {
+ "epoch": 1.2868921138184133,
+ "grad_norm": 0.2925741374492645,
+ "learning_rate": 5.3505361930294905e-06,
+ "loss": 0.5131395816802978,
+ "step": 4800
+ },
+ {
+ "epoch": 1.29225458323558,
+ "grad_norm": 0.18971531093120575,
+ "learning_rate": 5.310321715817694e-06,
+ "loss": 0.455674409866333,
+ "step": 4820
+ },
+ {
+ "epoch": 1.2976170526527466,
+ "grad_norm": 0.16778405010700226,
+ "learning_rate": 5.270107238605898e-06,
+ "loss": 0.5070962905883789,
+ "step": 4840
+ },
+ {
+ "epoch": 1.3029795220699132,
+ "grad_norm": 0.30026957392692566,
+ "learning_rate": 5.2298927613941016e-06,
+ "loss": 0.5120027542114258,
+ "step": 4860
+ },
+ {
+ "epoch": 1.3083419914870797,
+ "grad_norm": 0.17846634984016418,
+ "learning_rate": 5.189678284182305e-06,
+ "loss": 0.5114477157592774,
+ "step": 4880
+ },
+ {
+ "epoch": 1.3137044609042463,
+ "grad_norm": 0.1962418258190155,
+ "learning_rate": 5.149463806970509e-06,
+ "loss": 0.5043613910675049,
+ "step": 4900
+ },
+ {
+ "epoch": 1.319066930321413,
+ "grad_norm": 0.18446756899356842,
+ "learning_rate": 5.1092493297587135e-06,
+ "loss": 0.5396455287933349,
+ "step": 4920
+ },
+ {
+ "epoch": 1.3244293997385796,
+ "grad_norm": 0.20886844396591187,
+ "learning_rate": 5.069034852546917e-06,
+ "loss": 0.4879767417907715,
+ "step": 4940
+ },
+ {
+ "epoch": 1.3297918691557462,
+ "grad_norm": 0.16687901318073273,
+ "learning_rate": 5.028820375335121e-06,
+ "loss": 0.5014327049255372,
+ "step": 4960
+ },
+ {
+ "epoch": 1.3351543385729128,
+ "grad_norm": 0.19595153629779816,
+ "learning_rate": 4.988605898123325e-06,
+ "loss": 0.5375277996063232,
+ "step": 4980
+ },
+ {
+ "epoch": 1.3405168079900793,
+ "grad_norm": 0.2372344732284546,
+ "learning_rate": 4.948391420911528e-06,
+ "loss": 0.5020076274871826,
+ "step": 5000
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 6.159390743012475e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-5000/training_args.bin b/checkpoint-5000/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-5000/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-5200/README.md b/checkpoint-5200/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-5200/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-5200/adapter_config.json b/checkpoint-5200/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-5200/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-5200/adapter_model.safetensors b/checkpoint-5200/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..35f592962753bc8966d03b6525c6a7744a9dec87
--- /dev/null
+++ b/checkpoint-5200/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:146eef7222fc9494a8bc761b535e0037457f0a128ba28be257ebf7b38dc38ff5
+size 50360752
diff --git a/checkpoint-5200/chat_template.jinja b/checkpoint-5200/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-5200/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-5200/optimizer.pt b/checkpoint-5200/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..2372dd8ef1822cee7c2b846881b93d5ff2dccc96
--- /dev/null
+++ b/checkpoint-5200/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:de3b4bdc043543563bce989493312b7726f9e7e060546c4cb56c159edeab589b
+size 100828235
diff --git a/checkpoint-5200/rng_state.pth b/checkpoint-5200/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..f3e67e6ade4cb19889ae474828386f05acfbef58
--- /dev/null
+++ b/checkpoint-5200/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:632c73c4e7c60774a62a54d09cb677cc6262cb44d7e9b66d5cf59806414ad8cc
+size 14645
diff --git a/checkpoint-5200/scheduler.pt b/checkpoint-5200/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..d452350fcd1303c4b911fe507a0241a6e3d842c7
--- /dev/null
+++ b/checkpoint-5200/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:689a975460e3fd895001e99f36632cb4af3512fd358dd1d8ee5e76dbdbaca867
+size 1465
diff --git a/checkpoint-5200/tokenizer.json b/checkpoint-5200/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-5200/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-5200/tokenizer_config.json b/checkpoint-5200/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-5200/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-5200/trainer_state.json b/checkpoint-5200/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..c5c73620d2deffad18ef3f05179b6ae0376576e4
--- /dev/null
+++ b/checkpoint-5200/trainer_state.json
@@ -0,0 +1,1854 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.3941415021617454,
+ "eval_steps": 500,
+ "global_step": 5200,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ },
+ {
+ "epoch": 1.185005194892248,
+ "grad_norm": 0.16667844355106354,
+ "learning_rate": 6.114611260053619e-06,
+ "loss": 0.5159809589385986,
+ "step": 4420
+ },
+ {
+ "epoch": 1.1903676643094145,
+ "grad_norm": 0.19206570088863373,
+ "learning_rate": 6.074396782841823e-06,
+ "loss": 0.47541089057922364,
+ "step": 4440
+ },
+ {
+ "epoch": 1.195730133726581,
+ "grad_norm": 0.17394617199897766,
+ "learning_rate": 6.034182305630027e-06,
+ "loss": 0.5470661640167236,
+ "step": 4460
+ },
+ {
+ "epoch": 1.2010926031437477,
+ "grad_norm": 0.210404634475708,
+ "learning_rate": 5.993967828418231e-06,
+ "loss": 0.5377882957458496,
+ "step": 4480
+ },
+ {
+ "epoch": 1.2064550725609142,
+ "grad_norm": 0.18084648251533508,
+ "learning_rate": 5.953753351206435e-06,
+ "loss": 0.5037185192108155,
+ "step": 4500
+ },
+ {
+ "epoch": 1.211817541978081,
+ "grad_norm": 0.23707027733325958,
+ "learning_rate": 5.913538873994639e-06,
+ "loss": 0.4822190284729004,
+ "step": 4520
+ },
+ {
+ "epoch": 1.2171800113952476,
+ "grad_norm": 0.16474473476409912,
+ "learning_rate": 5.873324396782842e-06,
+ "loss": 0.46645288467407225,
+ "step": 4540
+ },
+ {
+ "epoch": 1.2225424808124141,
+ "grad_norm": 0.2142348438501358,
+ "learning_rate": 5.833109919571046e-06,
+ "loss": 0.5255855560302735,
+ "step": 4560
+ },
+ {
+ "epoch": 1.2279049502295807,
+ "grad_norm": 0.2531765103340149,
+ "learning_rate": 5.79289544235925e-06,
+ "loss": 0.507044792175293,
+ "step": 4580
+ },
+ {
+ "epoch": 1.2332674196467472,
+ "grad_norm": 0.2553550899028778,
+ "learning_rate": 5.7526809651474535e-06,
+ "loss": 0.4767824649810791,
+ "step": 4600
+ },
+ {
+ "epoch": 1.238629889063914,
+ "grad_norm": 0.14484412968158722,
+ "learning_rate": 5.712466487935657e-06,
+ "loss": 0.4675601005554199,
+ "step": 4620
+ },
+ {
+ "epoch": 1.2439923584810806,
+ "grad_norm": 0.14328251779079437,
+ "learning_rate": 5.672252010723861e-06,
+ "loss": 0.4956005573272705,
+ "step": 4640
+ },
+ {
+ "epoch": 1.2493548278982471,
+ "grad_norm": 0.1739245355129242,
+ "learning_rate": 5.632037533512065e-06,
+ "loss": 0.48583345413208007,
+ "step": 4660
+ },
+ {
+ "epoch": 1.2547172973154137,
+ "grad_norm": 0.21294184029102325,
+ "learning_rate": 5.591823056300268e-06,
+ "loss": 0.520921277999878,
+ "step": 4680
+ },
+ {
+ "epoch": 1.2600797667325803,
+ "grad_norm": 0.25132355093955994,
+ "learning_rate": 5.551608579088472e-06,
+ "loss": 0.5295385837554931,
+ "step": 4700
+ },
+ {
+ "epoch": 1.265442236149747,
+ "grad_norm": 0.18603841960430145,
+ "learning_rate": 5.511394101876676e-06,
+ "loss": 0.47570199966430665,
+ "step": 4720
+ },
+ {
+ "epoch": 1.2708047055669136,
+ "grad_norm": 0.19883134961128235,
+ "learning_rate": 5.471179624664879e-06,
+ "loss": 0.5016080379486084,
+ "step": 4740
+ },
+ {
+ "epoch": 1.2761671749840802,
+ "grad_norm": 0.19640181958675385,
+ "learning_rate": 5.430965147453083e-06,
+ "loss": 0.4999081134796143,
+ "step": 4760
+ },
+ {
+ "epoch": 1.2815296444012467,
+ "grad_norm": 0.2584764361381531,
+ "learning_rate": 5.390750670241287e-06,
+ "loss": 0.4780082702636719,
+ "step": 4780
+ },
+ {
+ "epoch": 1.2868921138184133,
+ "grad_norm": 0.2925741374492645,
+ "learning_rate": 5.3505361930294905e-06,
+ "loss": 0.5131395816802978,
+ "step": 4800
+ },
+ {
+ "epoch": 1.29225458323558,
+ "grad_norm": 0.18971531093120575,
+ "learning_rate": 5.310321715817694e-06,
+ "loss": 0.455674409866333,
+ "step": 4820
+ },
+ {
+ "epoch": 1.2976170526527466,
+ "grad_norm": 0.16778405010700226,
+ "learning_rate": 5.270107238605898e-06,
+ "loss": 0.5070962905883789,
+ "step": 4840
+ },
+ {
+ "epoch": 1.3029795220699132,
+ "grad_norm": 0.30026957392692566,
+ "learning_rate": 5.2298927613941016e-06,
+ "loss": 0.5120027542114258,
+ "step": 4860
+ },
+ {
+ "epoch": 1.3083419914870797,
+ "grad_norm": 0.17846634984016418,
+ "learning_rate": 5.189678284182305e-06,
+ "loss": 0.5114477157592774,
+ "step": 4880
+ },
+ {
+ "epoch": 1.3137044609042463,
+ "grad_norm": 0.1962418258190155,
+ "learning_rate": 5.149463806970509e-06,
+ "loss": 0.5043613910675049,
+ "step": 4900
+ },
+ {
+ "epoch": 1.319066930321413,
+ "grad_norm": 0.18446756899356842,
+ "learning_rate": 5.1092493297587135e-06,
+ "loss": 0.5396455287933349,
+ "step": 4920
+ },
+ {
+ "epoch": 1.3244293997385796,
+ "grad_norm": 0.20886844396591187,
+ "learning_rate": 5.069034852546917e-06,
+ "loss": 0.4879767417907715,
+ "step": 4940
+ },
+ {
+ "epoch": 1.3297918691557462,
+ "grad_norm": 0.16687901318073273,
+ "learning_rate": 5.028820375335121e-06,
+ "loss": 0.5014327049255372,
+ "step": 4960
+ },
+ {
+ "epoch": 1.3351543385729128,
+ "grad_norm": 0.19595153629779816,
+ "learning_rate": 4.988605898123325e-06,
+ "loss": 0.5375277996063232,
+ "step": 4980
+ },
+ {
+ "epoch": 1.3405168079900793,
+ "grad_norm": 0.2372344732284546,
+ "learning_rate": 4.948391420911528e-06,
+ "loss": 0.5020076274871826,
+ "step": 5000
+ },
+ {
+ "epoch": 1.345879277407246,
+ "grad_norm": 0.21030014753341675,
+ "learning_rate": 4.908176943699732e-06,
+ "loss": 0.5111066818237304,
+ "step": 5020
+ },
+ {
+ "epoch": 1.3512417468244127,
+ "grad_norm": 0.1866692751646042,
+ "learning_rate": 4.867962466487936e-06,
+ "loss": 0.4515383720397949,
+ "step": 5040
+ },
+ {
+ "epoch": 1.3566042162415792,
+ "grad_norm": 0.22531798481941223,
+ "learning_rate": 4.827747989276139e-06,
+ "loss": 0.4757690906524658,
+ "step": 5060
+ },
+ {
+ "epoch": 1.3619666856587458,
+ "grad_norm": 0.15868768095970154,
+ "learning_rate": 4.787533512064343e-06,
+ "loss": 0.45842318534851073,
+ "step": 5080
+ },
+ {
+ "epoch": 1.3673291550759124,
+ "grad_norm": 0.24528546631336212,
+ "learning_rate": 4.747319034852547e-06,
+ "loss": 0.47269258499145506,
+ "step": 5100
+ },
+ {
+ "epoch": 1.3726916244930791,
+ "grad_norm": 0.17387732863426208,
+ "learning_rate": 4.707104557640751e-06,
+ "loss": 0.5103805065155029,
+ "step": 5120
+ },
+ {
+ "epoch": 1.3780540939102457,
+ "grad_norm": 0.20686905086040497,
+ "learning_rate": 4.666890080428955e-06,
+ "loss": 0.5135180950164795,
+ "step": 5140
+ },
+ {
+ "epoch": 1.3834165633274123,
+ "grad_norm": 0.19599783420562744,
+ "learning_rate": 4.626675603217159e-06,
+ "loss": 0.5045839786529541,
+ "step": 5160
+ },
+ {
+ "epoch": 1.3887790327445788,
+ "grad_norm": 0.2585010528564453,
+ "learning_rate": 4.586461126005362e-06,
+ "loss": 0.45903496742248534,
+ "step": 5180
+ },
+ {
+ "epoch": 1.3941415021617454,
+ "grad_norm": 0.1688319593667984,
+ "learning_rate": 4.546246648793566e-06,
+ "loss": 0.5017509937286377,
+ "step": 5200
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 6.407414492442685e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-5200/training_args.bin b/checkpoint-5200/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-5200/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-5400/README.md b/checkpoint-5400/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-5400/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-5400/adapter_config.json b/checkpoint-5400/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-5400/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-5400/adapter_model.safetensors b/checkpoint-5400/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..2eb4981d81b334c1b3f22817b2ed85fa808e40c8
--- /dev/null
+++ b/checkpoint-5400/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:28141b68025bab841580b56ce0468675165e7f78435de7dc1cb94d15a3148cd7
+size 50360752
diff --git a/checkpoint-5400/chat_template.jinja b/checkpoint-5400/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-5400/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-5400/optimizer.pt b/checkpoint-5400/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..3777ddaadc140a28e736e2477114b47abc2661a5
--- /dev/null
+++ b/checkpoint-5400/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0155b3df218cbedbd2627d9c8b99b5b0c375c119c891304d6db43641e80a0281
+size 100828235
diff --git a/checkpoint-5400/rng_state.pth b/checkpoint-5400/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..77ac9b767ad0fb17455128c37a9bc3ef30efc8bf
--- /dev/null
+++ b/checkpoint-5400/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:defa5f944fd5a2264e613ebd98def7c3d56b4346aef74846345ce7b7eb28580a
+size 14645
diff --git a/checkpoint-5400/scheduler.pt b/checkpoint-5400/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..9f5c790d78e0557fe9d583379cb31e3c56c6049d
--- /dev/null
+++ b/checkpoint-5400/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:adf338bfedeceeaa4c29010d52f250a866fed7a9ac20388b9a7570b82a22b7d3
+size 1465
diff --git a/checkpoint-5400/tokenizer.json b/checkpoint-5400/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-5400/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-5400/tokenizer_config.json b/checkpoint-5400/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-5400/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-5400/trainer_state.json b/checkpoint-5400/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..5595a1d764c05d693da620a109e276392e58bdd7
--- /dev/null
+++ b/checkpoint-5400/trainer_state.json
@@ -0,0 +1,1924 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.4477661963334114,
+ "eval_steps": 500,
+ "global_step": 5400,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ },
+ {
+ "epoch": 1.185005194892248,
+ "grad_norm": 0.16667844355106354,
+ "learning_rate": 6.114611260053619e-06,
+ "loss": 0.5159809589385986,
+ "step": 4420
+ },
+ {
+ "epoch": 1.1903676643094145,
+ "grad_norm": 0.19206570088863373,
+ "learning_rate": 6.074396782841823e-06,
+ "loss": 0.47541089057922364,
+ "step": 4440
+ },
+ {
+ "epoch": 1.195730133726581,
+ "grad_norm": 0.17394617199897766,
+ "learning_rate": 6.034182305630027e-06,
+ "loss": 0.5470661640167236,
+ "step": 4460
+ },
+ {
+ "epoch": 1.2010926031437477,
+ "grad_norm": 0.210404634475708,
+ "learning_rate": 5.993967828418231e-06,
+ "loss": 0.5377882957458496,
+ "step": 4480
+ },
+ {
+ "epoch": 1.2064550725609142,
+ "grad_norm": 0.18084648251533508,
+ "learning_rate": 5.953753351206435e-06,
+ "loss": 0.5037185192108155,
+ "step": 4500
+ },
+ {
+ "epoch": 1.211817541978081,
+ "grad_norm": 0.23707027733325958,
+ "learning_rate": 5.913538873994639e-06,
+ "loss": 0.4822190284729004,
+ "step": 4520
+ },
+ {
+ "epoch": 1.2171800113952476,
+ "grad_norm": 0.16474473476409912,
+ "learning_rate": 5.873324396782842e-06,
+ "loss": 0.46645288467407225,
+ "step": 4540
+ },
+ {
+ "epoch": 1.2225424808124141,
+ "grad_norm": 0.2142348438501358,
+ "learning_rate": 5.833109919571046e-06,
+ "loss": 0.5255855560302735,
+ "step": 4560
+ },
+ {
+ "epoch": 1.2279049502295807,
+ "grad_norm": 0.2531765103340149,
+ "learning_rate": 5.79289544235925e-06,
+ "loss": 0.507044792175293,
+ "step": 4580
+ },
+ {
+ "epoch": 1.2332674196467472,
+ "grad_norm": 0.2553550899028778,
+ "learning_rate": 5.7526809651474535e-06,
+ "loss": 0.4767824649810791,
+ "step": 4600
+ },
+ {
+ "epoch": 1.238629889063914,
+ "grad_norm": 0.14484412968158722,
+ "learning_rate": 5.712466487935657e-06,
+ "loss": 0.4675601005554199,
+ "step": 4620
+ },
+ {
+ "epoch": 1.2439923584810806,
+ "grad_norm": 0.14328251779079437,
+ "learning_rate": 5.672252010723861e-06,
+ "loss": 0.4956005573272705,
+ "step": 4640
+ },
+ {
+ "epoch": 1.2493548278982471,
+ "grad_norm": 0.1739245355129242,
+ "learning_rate": 5.632037533512065e-06,
+ "loss": 0.48583345413208007,
+ "step": 4660
+ },
+ {
+ "epoch": 1.2547172973154137,
+ "grad_norm": 0.21294184029102325,
+ "learning_rate": 5.591823056300268e-06,
+ "loss": 0.520921277999878,
+ "step": 4680
+ },
+ {
+ "epoch": 1.2600797667325803,
+ "grad_norm": 0.25132355093955994,
+ "learning_rate": 5.551608579088472e-06,
+ "loss": 0.5295385837554931,
+ "step": 4700
+ },
+ {
+ "epoch": 1.265442236149747,
+ "grad_norm": 0.18603841960430145,
+ "learning_rate": 5.511394101876676e-06,
+ "loss": 0.47570199966430665,
+ "step": 4720
+ },
+ {
+ "epoch": 1.2708047055669136,
+ "grad_norm": 0.19883134961128235,
+ "learning_rate": 5.471179624664879e-06,
+ "loss": 0.5016080379486084,
+ "step": 4740
+ },
+ {
+ "epoch": 1.2761671749840802,
+ "grad_norm": 0.19640181958675385,
+ "learning_rate": 5.430965147453083e-06,
+ "loss": 0.4999081134796143,
+ "step": 4760
+ },
+ {
+ "epoch": 1.2815296444012467,
+ "grad_norm": 0.2584764361381531,
+ "learning_rate": 5.390750670241287e-06,
+ "loss": 0.4780082702636719,
+ "step": 4780
+ },
+ {
+ "epoch": 1.2868921138184133,
+ "grad_norm": 0.2925741374492645,
+ "learning_rate": 5.3505361930294905e-06,
+ "loss": 0.5131395816802978,
+ "step": 4800
+ },
+ {
+ "epoch": 1.29225458323558,
+ "grad_norm": 0.18971531093120575,
+ "learning_rate": 5.310321715817694e-06,
+ "loss": 0.455674409866333,
+ "step": 4820
+ },
+ {
+ "epoch": 1.2976170526527466,
+ "grad_norm": 0.16778405010700226,
+ "learning_rate": 5.270107238605898e-06,
+ "loss": 0.5070962905883789,
+ "step": 4840
+ },
+ {
+ "epoch": 1.3029795220699132,
+ "grad_norm": 0.30026957392692566,
+ "learning_rate": 5.2298927613941016e-06,
+ "loss": 0.5120027542114258,
+ "step": 4860
+ },
+ {
+ "epoch": 1.3083419914870797,
+ "grad_norm": 0.17846634984016418,
+ "learning_rate": 5.189678284182305e-06,
+ "loss": 0.5114477157592774,
+ "step": 4880
+ },
+ {
+ "epoch": 1.3137044609042463,
+ "grad_norm": 0.1962418258190155,
+ "learning_rate": 5.149463806970509e-06,
+ "loss": 0.5043613910675049,
+ "step": 4900
+ },
+ {
+ "epoch": 1.319066930321413,
+ "grad_norm": 0.18446756899356842,
+ "learning_rate": 5.1092493297587135e-06,
+ "loss": 0.5396455287933349,
+ "step": 4920
+ },
+ {
+ "epoch": 1.3244293997385796,
+ "grad_norm": 0.20886844396591187,
+ "learning_rate": 5.069034852546917e-06,
+ "loss": 0.4879767417907715,
+ "step": 4940
+ },
+ {
+ "epoch": 1.3297918691557462,
+ "grad_norm": 0.16687901318073273,
+ "learning_rate": 5.028820375335121e-06,
+ "loss": 0.5014327049255372,
+ "step": 4960
+ },
+ {
+ "epoch": 1.3351543385729128,
+ "grad_norm": 0.19595153629779816,
+ "learning_rate": 4.988605898123325e-06,
+ "loss": 0.5375277996063232,
+ "step": 4980
+ },
+ {
+ "epoch": 1.3405168079900793,
+ "grad_norm": 0.2372344732284546,
+ "learning_rate": 4.948391420911528e-06,
+ "loss": 0.5020076274871826,
+ "step": 5000
+ },
+ {
+ "epoch": 1.345879277407246,
+ "grad_norm": 0.21030014753341675,
+ "learning_rate": 4.908176943699732e-06,
+ "loss": 0.5111066818237304,
+ "step": 5020
+ },
+ {
+ "epoch": 1.3512417468244127,
+ "grad_norm": 0.1866692751646042,
+ "learning_rate": 4.867962466487936e-06,
+ "loss": 0.4515383720397949,
+ "step": 5040
+ },
+ {
+ "epoch": 1.3566042162415792,
+ "grad_norm": 0.22531798481941223,
+ "learning_rate": 4.827747989276139e-06,
+ "loss": 0.4757690906524658,
+ "step": 5060
+ },
+ {
+ "epoch": 1.3619666856587458,
+ "grad_norm": 0.15868768095970154,
+ "learning_rate": 4.787533512064343e-06,
+ "loss": 0.45842318534851073,
+ "step": 5080
+ },
+ {
+ "epoch": 1.3673291550759124,
+ "grad_norm": 0.24528546631336212,
+ "learning_rate": 4.747319034852547e-06,
+ "loss": 0.47269258499145506,
+ "step": 5100
+ },
+ {
+ "epoch": 1.3726916244930791,
+ "grad_norm": 0.17387732863426208,
+ "learning_rate": 4.707104557640751e-06,
+ "loss": 0.5103805065155029,
+ "step": 5120
+ },
+ {
+ "epoch": 1.3780540939102457,
+ "grad_norm": 0.20686905086040497,
+ "learning_rate": 4.666890080428955e-06,
+ "loss": 0.5135180950164795,
+ "step": 5140
+ },
+ {
+ "epoch": 1.3834165633274123,
+ "grad_norm": 0.19599783420562744,
+ "learning_rate": 4.626675603217159e-06,
+ "loss": 0.5045839786529541,
+ "step": 5160
+ },
+ {
+ "epoch": 1.3887790327445788,
+ "grad_norm": 0.2585010528564453,
+ "learning_rate": 4.586461126005362e-06,
+ "loss": 0.45903496742248534,
+ "step": 5180
+ },
+ {
+ "epoch": 1.3941415021617454,
+ "grad_norm": 0.1688319593667984,
+ "learning_rate": 4.546246648793566e-06,
+ "loss": 0.5017509937286377,
+ "step": 5200
+ },
+ {
+ "epoch": 1.3995039715789122,
+ "grad_norm": 0.21520815789699554,
+ "learning_rate": 4.50603217158177e-06,
+ "loss": 0.48459539413452146,
+ "step": 5220
+ },
+ {
+ "epoch": 1.4048664409960787,
+ "grad_norm": 0.20514647662639618,
+ "learning_rate": 4.4658176943699735e-06,
+ "loss": 0.5073423862457276,
+ "step": 5240
+ },
+ {
+ "epoch": 1.4102289104132453,
+ "grad_norm": 0.21835413575172424,
+ "learning_rate": 4.425603217158177e-06,
+ "loss": 0.5290310382843018,
+ "step": 5260
+ },
+ {
+ "epoch": 1.4155913798304118,
+ "grad_norm": 0.28042587637901306,
+ "learning_rate": 4.385388739946381e-06,
+ "loss": 0.4823312759399414,
+ "step": 5280
+ },
+ {
+ "epoch": 1.4209538492475784,
+ "grad_norm": 0.18959026038646698,
+ "learning_rate": 4.345174262734585e-06,
+ "loss": 0.4921241760253906,
+ "step": 5300
+ },
+ {
+ "epoch": 1.4263163186647452,
+ "grad_norm": 0.18584316968917847,
+ "learning_rate": 4.304959785522788e-06,
+ "loss": 0.4892130374908447,
+ "step": 5320
+ },
+ {
+ "epoch": 1.4316787880819117,
+ "grad_norm": 0.17588038742542267,
+ "learning_rate": 4.264745308310992e-06,
+ "loss": 0.4822041988372803,
+ "step": 5340
+ },
+ {
+ "epoch": 1.4370412574990783,
+ "grad_norm": 0.18146033585071564,
+ "learning_rate": 4.224530831099196e-06,
+ "loss": 0.5084807395935058,
+ "step": 5360
+ },
+ {
+ "epoch": 1.4424037269162449,
+ "grad_norm": 0.2251797467470169,
+ "learning_rate": 4.184316353887399e-06,
+ "loss": 0.5146170139312745,
+ "step": 5380
+ },
+ {
+ "epoch": 1.4477661963334114,
+ "grad_norm": 0.18744796514511108,
+ "learning_rate": 4.144101876675603e-06,
+ "loss": 0.5189927577972412,
+ "step": 5400
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 6.652352029577196e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-5400/training_args.bin b/checkpoint-5400/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-5400/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-5600/README.md b/checkpoint-5600/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-5600/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-5600/adapter_config.json b/checkpoint-5600/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-5600/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-5600/adapter_model.safetensors b/checkpoint-5600/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..6da8258121afcff7481ebd61dbc536630a39b97d
--- /dev/null
+++ b/checkpoint-5600/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3ea4e2bcd387e044005e814063b1225dbfc8890aa509e03eb1431bf2d3644ea9
+size 50360752
diff --git a/checkpoint-5600/chat_template.jinja b/checkpoint-5600/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-5600/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-5600/optimizer.pt b/checkpoint-5600/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..6db6396492016f102fa451cebb4b171354daad8c
--- /dev/null
+++ b/checkpoint-5600/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:be14ed84978828c6cc9295376bb460c350d5ac0806608f3458fa1c7a039fdee0
+size 100828235
diff --git a/checkpoint-5600/rng_state.pth b/checkpoint-5600/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..4368707c059765bc10ef490ba129ad99bb506e57
--- /dev/null
+++ b/checkpoint-5600/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5791c2a0d1c09d1732c89ce549a6e832485b37524facaa2642c76169e19f6d1b
+size 14645
diff --git a/checkpoint-5600/scheduler.pt b/checkpoint-5600/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..f980c2969f5c2b349c98715d6aa548a3fd19dea8
--- /dev/null
+++ b/checkpoint-5600/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:56f6cace1076b12bf8584bd4d77224683396a0269b67fae379ebc20bbe096585
+size 1465
diff --git a/checkpoint-5600/tokenizer.json b/checkpoint-5600/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-5600/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-5600/tokenizer_config.json b/checkpoint-5600/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-5600/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-5600/trainer_state.json b/checkpoint-5600/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..d8f8c3643238fd2be19c40f080d5cb354693d286
--- /dev/null
+++ b/checkpoint-5600/trainer_state.json
@@ -0,0 +1,1994 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.5013908905050775,
+ "eval_steps": 500,
+ "global_step": 5600,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ },
+ {
+ "epoch": 1.185005194892248,
+ "grad_norm": 0.16667844355106354,
+ "learning_rate": 6.114611260053619e-06,
+ "loss": 0.5159809589385986,
+ "step": 4420
+ },
+ {
+ "epoch": 1.1903676643094145,
+ "grad_norm": 0.19206570088863373,
+ "learning_rate": 6.074396782841823e-06,
+ "loss": 0.47541089057922364,
+ "step": 4440
+ },
+ {
+ "epoch": 1.195730133726581,
+ "grad_norm": 0.17394617199897766,
+ "learning_rate": 6.034182305630027e-06,
+ "loss": 0.5470661640167236,
+ "step": 4460
+ },
+ {
+ "epoch": 1.2010926031437477,
+ "grad_norm": 0.210404634475708,
+ "learning_rate": 5.993967828418231e-06,
+ "loss": 0.5377882957458496,
+ "step": 4480
+ },
+ {
+ "epoch": 1.2064550725609142,
+ "grad_norm": 0.18084648251533508,
+ "learning_rate": 5.953753351206435e-06,
+ "loss": 0.5037185192108155,
+ "step": 4500
+ },
+ {
+ "epoch": 1.211817541978081,
+ "grad_norm": 0.23707027733325958,
+ "learning_rate": 5.913538873994639e-06,
+ "loss": 0.4822190284729004,
+ "step": 4520
+ },
+ {
+ "epoch": 1.2171800113952476,
+ "grad_norm": 0.16474473476409912,
+ "learning_rate": 5.873324396782842e-06,
+ "loss": 0.46645288467407225,
+ "step": 4540
+ },
+ {
+ "epoch": 1.2225424808124141,
+ "grad_norm": 0.2142348438501358,
+ "learning_rate": 5.833109919571046e-06,
+ "loss": 0.5255855560302735,
+ "step": 4560
+ },
+ {
+ "epoch": 1.2279049502295807,
+ "grad_norm": 0.2531765103340149,
+ "learning_rate": 5.79289544235925e-06,
+ "loss": 0.507044792175293,
+ "step": 4580
+ },
+ {
+ "epoch": 1.2332674196467472,
+ "grad_norm": 0.2553550899028778,
+ "learning_rate": 5.7526809651474535e-06,
+ "loss": 0.4767824649810791,
+ "step": 4600
+ },
+ {
+ "epoch": 1.238629889063914,
+ "grad_norm": 0.14484412968158722,
+ "learning_rate": 5.712466487935657e-06,
+ "loss": 0.4675601005554199,
+ "step": 4620
+ },
+ {
+ "epoch": 1.2439923584810806,
+ "grad_norm": 0.14328251779079437,
+ "learning_rate": 5.672252010723861e-06,
+ "loss": 0.4956005573272705,
+ "step": 4640
+ },
+ {
+ "epoch": 1.2493548278982471,
+ "grad_norm": 0.1739245355129242,
+ "learning_rate": 5.632037533512065e-06,
+ "loss": 0.48583345413208007,
+ "step": 4660
+ },
+ {
+ "epoch": 1.2547172973154137,
+ "grad_norm": 0.21294184029102325,
+ "learning_rate": 5.591823056300268e-06,
+ "loss": 0.520921277999878,
+ "step": 4680
+ },
+ {
+ "epoch": 1.2600797667325803,
+ "grad_norm": 0.25132355093955994,
+ "learning_rate": 5.551608579088472e-06,
+ "loss": 0.5295385837554931,
+ "step": 4700
+ },
+ {
+ "epoch": 1.265442236149747,
+ "grad_norm": 0.18603841960430145,
+ "learning_rate": 5.511394101876676e-06,
+ "loss": 0.47570199966430665,
+ "step": 4720
+ },
+ {
+ "epoch": 1.2708047055669136,
+ "grad_norm": 0.19883134961128235,
+ "learning_rate": 5.471179624664879e-06,
+ "loss": 0.5016080379486084,
+ "step": 4740
+ },
+ {
+ "epoch": 1.2761671749840802,
+ "grad_norm": 0.19640181958675385,
+ "learning_rate": 5.430965147453083e-06,
+ "loss": 0.4999081134796143,
+ "step": 4760
+ },
+ {
+ "epoch": 1.2815296444012467,
+ "grad_norm": 0.2584764361381531,
+ "learning_rate": 5.390750670241287e-06,
+ "loss": 0.4780082702636719,
+ "step": 4780
+ },
+ {
+ "epoch": 1.2868921138184133,
+ "grad_norm": 0.2925741374492645,
+ "learning_rate": 5.3505361930294905e-06,
+ "loss": 0.5131395816802978,
+ "step": 4800
+ },
+ {
+ "epoch": 1.29225458323558,
+ "grad_norm": 0.18971531093120575,
+ "learning_rate": 5.310321715817694e-06,
+ "loss": 0.455674409866333,
+ "step": 4820
+ },
+ {
+ "epoch": 1.2976170526527466,
+ "grad_norm": 0.16778405010700226,
+ "learning_rate": 5.270107238605898e-06,
+ "loss": 0.5070962905883789,
+ "step": 4840
+ },
+ {
+ "epoch": 1.3029795220699132,
+ "grad_norm": 0.30026957392692566,
+ "learning_rate": 5.2298927613941016e-06,
+ "loss": 0.5120027542114258,
+ "step": 4860
+ },
+ {
+ "epoch": 1.3083419914870797,
+ "grad_norm": 0.17846634984016418,
+ "learning_rate": 5.189678284182305e-06,
+ "loss": 0.5114477157592774,
+ "step": 4880
+ },
+ {
+ "epoch": 1.3137044609042463,
+ "grad_norm": 0.1962418258190155,
+ "learning_rate": 5.149463806970509e-06,
+ "loss": 0.5043613910675049,
+ "step": 4900
+ },
+ {
+ "epoch": 1.319066930321413,
+ "grad_norm": 0.18446756899356842,
+ "learning_rate": 5.1092493297587135e-06,
+ "loss": 0.5396455287933349,
+ "step": 4920
+ },
+ {
+ "epoch": 1.3244293997385796,
+ "grad_norm": 0.20886844396591187,
+ "learning_rate": 5.069034852546917e-06,
+ "loss": 0.4879767417907715,
+ "step": 4940
+ },
+ {
+ "epoch": 1.3297918691557462,
+ "grad_norm": 0.16687901318073273,
+ "learning_rate": 5.028820375335121e-06,
+ "loss": 0.5014327049255372,
+ "step": 4960
+ },
+ {
+ "epoch": 1.3351543385729128,
+ "grad_norm": 0.19595153629779816,
+ "learning_rate": 4.988605898123325e-06,
+ "loss": 0.5375277996063232,
+ "step": 4980
+ },
+ {
+ "epoch": 1.3405168079900793,
+ "grad_norm": 0.2372344732284546,
+ "learning_rate": 4.948391420911528e-06,
+ "loss": 0.5020076274871826,
+ "step": 5000
+ },
+ {
+ "epoch": 1.345879277407246,
+ "grad_norm": 0.21030014753341675,
+ "learning_rate": 4.908176943699732e-06,
+ "loss": 0.5111066818237304,
+ "step": 5020
+ },
+ {
+ "epoch": 1.3512417468244127,
+ "grad_norm": 0.1866692751646042,
+ "learning_rate": 4.867962466487936e-06,
+ "loss": 0.4515383720397949,
+ "step": 5040
+ },
+ {
+ "epoch": 1.3566042162415792,
+ "grad_norm": 0.22531798481941223,
+ "learning_rate": 4.827747989276139e-06,
+ "loss": 0.4757690906524658,
+ "step": 5060
+ },
+ {
+ "epoch": 1.3619666856587458,
+ "grad_norm": 0.15868768095970154,
+ "learning_rate": 4.787533512064343e-06,
+ "loss": 0.45842318534851073,
+ "step": 5080
+ },
+ {
+ "epoch": 1.3673291550759124,
+ "grad_norm": 0.24528546631336212,
+ "learning_rate": 4.747319034852547e-06,
+ "loss": 0.47269258499145506,
+ "step": 5100
+ },
+ {
+ "epoch": 1.3726916244930791,
+ "grad_norm": 0.17387732863426208,
+ "learning_rate": 4.707104557640751e-06,
+ "loss": 0.5103805065155029,
+ "step": 5120
+ },
+ {
+ "epoch": 1.3780540939102457,
+ "grad_norm": 0.20686905086040497,
+ "learning_rate": 4.666890080428955e-06,
+ "loss": 0.5135180950164795,
+ "step": 5140
+ },
+ {
+ "epoch": 1.3834165633274123,
+ "grad_norm": 0.19599783420562744,
+ "learning_rate": 4.626675603217159e-06,
+ "loss": 0.5045839786529541,
+ "step": 5160
+ },
+ {
+ "epoch": 1.3887790327445788,
+ "grad_norm": 0.2585010528564453,
+ "learning_rate": 4.586461126005362e-06,
+ "loss": 0.45903496742248534,
+ "step": 5180
+ },
+ {
+ "epoch": 1.3941415021617454,
+ "grad_norm": 0.1688319593667984,
+ "learning_rate": 4.546246648793566e-06,
+ "loss": 0.5017509937286377,
+ "step": 5200
+ },
+ {
+ "epoch": 1.3995039715789122,
+ "grad_norm": 0.21520815789699554,
+ "learning_rate": 4.50603217158177e-06,
+ "loss": 0.48459539413452146,
+ "step": 5220
+ },
+ {
+ "epoch": 1.4048664409960787,
+ "grad_norm": 0.20514647662639618,
+ "learning_rate": 4.4658176943699735e-06,
+ "loss": 0.5073423862457276,
+ "step": 5240
+ },
+ {
+ "epoch": 1.4102289104132453,
+ "grad_norm": 0.21835413575172424,
+ "learning_rate": 4.425603217158177e-06,
+ "loss": 0.5290310382843018,
+ "step": 5260
+ },
+ {
+ "epoch": 1.4155913798304118,
+ "grad_norm": 0.28042587637901306,
+ "learning_rate": 4.385388739946381e-06,
+ "loss": 0.4823312759399414,
+ "step": 5280
+ },
+ {
+ "epoch": 1.4209538492475784,
+ "grad_norm": 0.18959026038646698,
+ "learning_rate": 4.345174262734585e-06,
+ "loss": 0.4921241760253906,
+ "step": 5300
+ },
+ {
+ "epoch": 1.4263163186647452,
+ "grad_norm": 0.18584316968917847,
+ "learning_rate": 4.304959785522788e-06,
+ "loss": 0.4892130374908447,
+ "step": 5320
+ },
+ {
+ "epoch": 1.4316787880819117,
+ "grad_norm": 0.17588038742542267,
+ "learning_rate": 4.264745308310992e-06,
+ "loss": 0.4822041988372803,
+ "step": 5340
+ },
+ {
+ "epoch": 1.4370412574990783,
+ "grad_norm": 0.18146033585071564,
+ "learning_rate": 4.224530831099196e-06,
+ "loss": 0.5084807395935058,
+ "step": 5360
+ },
+ {
+ "epoch": 1.4424037269162449,
+ "grad_norm": 0.2251797467470169,
+ "learning_rate": 4.184316353887399e-06,
+ "loss": 0.5146170139312745,
+ "step": 5380
+ },
+ {
+ "epoch": 1.4477661963334114,
+ "grad_norm": 0.18744796514511108,
+ "learning_rate": 4.144101876675603e-06,
+ "loss": 0.5189927577972412,
+ "step": 5400
+ },
+ {
+ "epoch": 1.4531286657505782,
+ "grad_norm": 0.25737133622169495,
+ "learning_rate": 4.103887399463807e-06,
+ "loss": 0.4891658782958984,
+ "step": 5420
+ },
+ {
+ "epoch": 1.4584911351677448,
+ "grad_norm": 0.20580479502677917,
+ "learning_rate": 4.0636729222520105e-06,
+ "loss": 0.4953591823577881,
+ "step": 5440
+ },
+ {
+ "epoch": 1.4638536045849113,
+ "grad_norm": 0.2351546287536621,
+ "learning_rate": 4.023458445040214e-06,
+ "loss": 0.5025320053100586,
+ "step": 5460
+ },
+ {
+ "epoch": 1.4692160740020779,
+ "grad_norm": 0.1819481998682022,
+ "learning_rate": 3.983243967828418e-06,
+ "loss": 0.47151756286621094,
+ "step": 5480
+ },
+ {
+ "epoch": 1.4745785434192444,
+ "grad_norm": 0.20772472023963928,
+ "learning_rate": 3.943029490616622e-06,
+ "loss": 0.4678915023803711,
+ "step": 5500
+ },
+ {
+ "epoch": 1.4799410128364112,
+ "grad_norm": 0.2203037440776825,
+ "learning_rate": 3.902815013404825e-06,
+ "loss": 0.46007452011108396,
+ "step": 5520
+ },
+ {
+ "epoch": 1.4853034822535778,
+ "grad_norm": 0.15371400117874146,
+ "learning_rate": 3.86260053619303e-06,
+ "loss": 0.44407024383544924,
+ "step": 5540
+ },
+ {
+ "epoch": 1.4906659516707443,
+ "grad_norm": 0.2276080846786499,
+ "learning_rate": 3.8223860589812335e-06,
+ "loss": 0.4730556488037109,
+ "step": 5560
+ },
+ {
+ "epoch": 1.4960284210879111,
+ "grad_norm": 0.24482466280460358,
+ "learning_rate": 3.7821715817694376e-06,
+ "loss": 0.5073911666870117,
+ "step": 5580
+ },
+ {
+ "epoch": 1.5013908905050775,
+ "grad_norm": 0.20438458025455475,
+ "learning_rate": 3.741957104557641e-06,
+ "loss": 0.46701641082763673,
+ "step": 5600
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 6.895036875406295e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-5600/training_args.bin b/checkpoint-5600/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-5600/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-5800/README.md b/checkpoint-5800/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-5800/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-5800/adapter_config.json b/checkpoint-5800/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-5800/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-5800/adapter_model.safetensors b/checkpoint-5800/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..a403e364d512b47cd8cae7d992e2ef182d8128d9
--- /dev/null
+++ b/checkpoint-5800/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:48a6c661168045f149a8bf9fc70b56e1a98a50603045f9a0b0319be5ef6c3b8a
+size 50360752
diff --git a/checkpoint-5800/chat_template.jinja b/checkpoint-5800/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-5800/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-5800/optimizer.pt b/checkpoint-5800/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..3579dc7b1d282e75ad6f46fa6b4f37c59ffd469a
--- /dev/null
+++ b/checkpoint-5800/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:cf8b53f5878689f0298adf6edbc474223bdf44d344afd820c162f6f48104bc98
+size 100828235
diff --git a/checkpoint-5800/rng_state.pth b/checkpoint-5800/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..528271c23234b9bc5137a624b9af989c64064949
--- /dev/null
+++ b/checkpoint-5800/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b98fa80b2bfcc9eb38e63a48bdf8e54218d3a8d41b0da0c5eaf3276e952ed7f5
+size 14645
diff --git a/checkpoint-5800/scheduler.pt b/checkpoint-5800/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..73c36f4c111d4a502f43f67a4f83e0c83fa011cb
--- /dev/null
+++ b/checkpoint-5800/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ceac609cbebd0e97bb85a708c488e23e04b9da9d0b459db394ef36def3d7040d
+size 1465
diff --git a/checkpoint-5800/tokenizer.json b/checkpoint-5800/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-5800/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-5800/tokenizer_config.json b/checkpoint-5800/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-5800/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-5800/trainer_state.json b/checkpoint-5800/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..2d285985ba9e8a933277746d56d395f1075304cd
--- /dev/null
+++ b/checkpoint-5800/trainer_state.json
@@ -0,0 +1,2064 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.5550155846767435,
+ "eval_steps": 500,
+ "global_step": 5800,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ },
+ {
+ "epoch": 1.185005194892248,
+ "grad_norm": 0.16667844355106354,
+ "learning_rate": 6.114611260053619e-06,
+ "loss": 0.5159809589385986,
+ "step": 4420
+ },
+ {
+ "epoch": 1.1903676643094145,
+ "grad_norm": 0.19206570088863373,
+ "learning_rate": 6.074396782841823e-06,
+ "loss": 0.47541089057922364,
+ "step": 4440
+ },
+ {
+ "epoch": 1.195730133726581,
+ "grad_norm": 0.17394617199897766,
+ "learning_rate": 6.034182305630027e-06,
+ "loss": 0.5470661640167236,
+ "step": 4460
+ },
+ {
+ "epoch": 1.2010926031437477,
+ "grad_norm": 0.210404634475708,
+ "learning_rate": 5.993967828418231e-06,
+ "loss": 0.5377882957458496,
+ "step": 4480
+ },
+ {
+ "epoch": 1.2064550725609142,
+ "grad_norm": 0.18084648251533508,
+ "learning_rate": 5.953753351206435e-06,
+ "loss": 0.5037185192108155,
+ "step": 4500
+ },
+ {
+ "epoch": 1.211817541978081,
+ "grad_norm": 0.23707027733325958,
+ "learning_rate": 5.913538873994639e-06,
+ "loss": 0.4822190284729004,
+ "step": 4520
+ },
+ {
+ "epoch": 1.2171800113952476,
+ "grad_norm": 0.16474473476409912,
+ "learning_rate": 5.873324396782842e-06,
+ "loss": 0.46645288467407225,
+ "step": 4540
+ },
+ {
+ "epoch": 1.2225424808124141,
+ "grad_norm": 0.2142348438501358,
+ "learning_rate": 5.833109919571046e-06,
+ "loss": 0.5255855560302735,
+ "step": 4560
+ },
+ {
+ "epoch": 1.2279049502295807,
+ "grad_norm": 0.2531765103340149,
+ "learning_rate": 5.79289544235925e-06,
+ "loss": 0.507044792175293,
+ "step": 4580
+ },
+ {
+ "epoch": 1.2332674196467472,
+ "grad_norm": 0.2553550899028778,
+ "learning_rate": 5.7526809651474535e-06,
+ "loss": 0.4767824649810791,
+ "step": 4600
+ },
+ {
+ "epoch": 1.238629889063914,
+ "grad_norm": 0.14484412968158722,
+ "learning_rate": 5.712466487935657e-06,
+ "loss": 0.4675601005554199,
+ "step": 4620
+ },
+ {
+ "epoch": 1.2439923584810806,
+ "grad_norm": 0.14328251779079437,
+ "learning_rate": 5.672252010723861e-06,
+ "loss": 0.4956005573272705,
+ "step": 4640
+ },
+ {
+ "epoch": 1.2493548278982471,
+ "grad_norm": 0.1739245355129242,
+ "learning_rate": 5.632037533512065e-06,
+ "loss": 0.48583345413208007,
+ "step": 4660
+ },
+ {
+ "epoch": 1.2547172973154137,
+ "grad_norm": 0.21294184029102325,
+ "learning_rate": 5.591823056300268e-06,
+ "loss": 0.520921277999878,
+ "step": 4680
+ },
+ {
+ "epoch": 1.2600797667325803,
+ "grad_norm": 0.25132355093955994,
+ "learning_rate": 5.551608579088472e-06,
+ "loss": 0.5295385837554931,
+ "step": 4700
+ },
+ {
+ "epoch": 1.265442236149747,
+ "grad_norm": 0.18603841960430145,
+ "learning_rate": 5.511394101876676e-06,
+ "loss": 0.47570199966430665,
+ "step": 4720
+ },
+ {
+ "epoch": 1.2708047055669136,
+ "grad_norm": 0.19883134961128235,
+ "learning_rate": 5.471179624664879e-06,
+ "loss": 0.5016080379486084,
+ "step": 4740
+ },
+ {
+ "epoch": 1.2761671749840802,
+ "grad_norm": 0.19640181958675385,
+ "learning_rate": 5.430965147453083e-06,
+ "loss": 0.4999081134796143,
+ "step": 4760
+ },
+ {
+ "epoch": 1.2815296444012467,
+ "grad_norm": 0.2584764361381531,
+ "learning_rate": 5.390750670241287e-06,
+ "loss": 0.4780082702636719,
+ "step": 4780
+ },
+ {
+ "epoch": 1.2868921138184133,
+ "grad_norm": 0.2925741374492645,
+ "learning_rate": 5.3505361930294905e-06,
+ "loss": 0.5131395816802978,
+ "step": 4800
+ },
+ {
+ "epoch": 1.29225458323558,
+ "grad_norm": 0.18971531093120575,
+ "learning_rate": 5.310321715817694e-06,
+ "loss": 0.455674409866333,
+ "step": 4820
+ },
+ {
+ "epoch": 1.2976170526527466,
+ "grad_norm": 0.16778405010700226,
+ "learning_rate": 5.270107238605898e-06,
+ "loss": 0.5070962905883789,
+ "step": 4840
+ },
+ {
+ "epoch": 1.3029795220699132,
+ "grad_norm": 0.30026957392692566,
+ "learning_rate": 5.2298927613941016e-06,
+ "loss": 0.5120027542114258,
+ "step": 4860
+ },
+ {
+ "epoch": 1.3083419914870797,
+ "grad_norm": 0.17846634984016418,
+ "learning_rate": 5.189678284182305e-06,
+ "loss": 0.5114477157592774,
+ "step": 4880
+ },
+ {
+ "epoch": 1.3137044609042463,
+ "grad_norm": 0.1962418258190155,
+ "learning_rate": 5.149463806970509e-06,
+ "loss": 0.5043613910675049,
+ "step": 4900
+ },
+ {
+ "epoch": 1.319066930321413,
+ "grad_norm": 0.18446756899356842,
+ "learning_rate": 5.1092493297587135e-06,
+ "loss": 0.5396455287933349,
+ "step": 4920
+ },
+ {
+ "epoch": 1.3244293997385796,
+ "grad_norm": 0.20886844396591187,
+ "learning_rate": 5.069034852546917e-06,
+ "loss": 0.4879767417907715,
+ "step": 4940
+ },
+ {
+ "epoch": 1.3297918691557462,
+ "grad_norm": 0.16687901318073273,
+ "learning_rate": 5.028820375335121e-06,
+ "loss": 0.5014327049255372,
+ "step": 4960
+ },
+ {
+ "epoch": 1.3351543385729128,
+ "grad_norm": 0.19595153629779816,
+ "learning_rate": 4.988605898123325e-06,
+ "loss": 0.5375277996063232,
+ "step": 4980
+ },
+ {
+ "epoch": 1.3405168079900793,
+ "grad_norm": 0.2372344732284546,
+ "learning_rate": 4.948391420911528e-06,
+ "loss": 0.5020076274871826,
+ "step": 5000
+ },
+ {
+ "epoch": 1.345879277407246,
+ "grad_norm": 0.21030014753341675,
+ "learning_rate": 4.908176943699732e-06,
+ "loss": 0.5111066818237304,
+ "step": 5020
+ },
+ {
+ "epoch": 1.3512417468244127,
+ "grad_norm": 0.1866692751646042,
+ "learning_rate": 4.867962466487936e-06,
+ "loss": 0.4515383720397949,
+ "step": 5040
+ },
+ {
+ "epoch": 1.3566042162415792,
+ "grad_norm": 0.22531798481941223,
+ "learning_rate": 4.827747989276139e-06,
+ "loss": 0.4757690906524658,
+ "step": 5060
+ },
+ {
+ "epoch": 1.3619666856587458,
+ "grad_norm": 0.15868768095970154,
+ "learning_rate": 4.787533512064343e-06,
+ "loss": 0.45842318534851073,
+ "step": 5080
+ },
+ {
+ "epoch": 1.3673291550759124,
+ "grad_norm": 0.24528546631336212,
+ "learning_rate": 4.747319034852547e-06,
+ "loss": 0.47269258499145506,
+ "step": 5100
+ },
+ {
+ "epoch": 1.3726916244930791,
+ "grad_norm": 0.17387732863426208,
+ "learning_rate": 4.707104557640751e-06,
+ "loss": 0.5103805065155029,
+ "step": 5120
+ },
+ {
+ "epoch": 1.3780540939102457,
+ "grad_norm": 0.20686905086040497,
+ "learning_rate": 4.666890080428955e-06,
+ "loss": 0.5135180950164795,
+ "step": 5140
+ },
+ {
+ "epoch": 1.3834165633274123,
+ "grad_norm": 0.19599783420562744,
+ "learning_rate": 4.626675603217159e-06,
+ "loss": 0.5045839786529541,
+ "step": 5160
+ },
+ {
+ "epoch": 1.3887790327445788,
+ "grad_norm": 0.2585010528564453,
+ "learning_rate": 4.586461126005362e-06,
+ "loss": 0.45903496742248534,
+ "step": 5180
+ },
+ {
+ "epoch": 1.3941415021617454,
+ "grad_norm": 0.1688319593667984,
+ "learning_rate": 4.546246648793566e-06,
+ "loss": 0.5017509937286377,
+ "step": 5200
+ },
+ {
+ "epoch": 1.3995039715789122,
+ "grad_norm": 0.21520815789699554,
+ "learning_rate": 4.50603217158177e-06,
+ "loss": 0.48459539413452146,
+ "step": 5220
+ },
+ {
+ "epoch": 1.4048664409960787,
+ "grad_norm": 0.20514647662639618,
+ "learning_rate": 4.4658176943699735e-06,
+ "loss": 0.5073423862457276,
+ "step": 5240
+ },
+ {
+ "epoch": 1.4102289104132453,
+ "grad_norm": 0.21835413575172424,
+ "learning_rate": 4.425603217158177e-06,
+ "loss": 0.5290310382843018,
+ "step": 5260
+ },
+ {
+ "epoch": 1.4155913798304118,
+ "grad_norm": 0.28042587637901306,
+ "learning_rate": 4.385388739946381e-06,
+ "loss": 0.4823312759399414,
+ "step": 5280
+ },
+ {
+ "epoch": 1.4209538492475784,
+ "grad_norm": 0.18959026038646698,
+ "learning_rate": 4.345174262734585e-06,
+ "loss": 0.4921241760253906,
+ "step": 5300
+ },
+ {
+ "epoch": 1.4263163186647452,
+ "grad_norm": 0.18584316968917847,
+ "learning_rate": 4.304959785522788e-06,
+ "loss": 0.4892130374908447,
+ "step": 5320
+ },
+ {
+ "epoch": 1.4316787880819117,
+ "grad_norm": 0.17588038742542267,
+ "learning_rate": 4.264745308310992e-06,
+ "loss": 0.4822041988372803,
+ "step": 5340
+ },
+ {
+ "epoch": 1.4370412574990783,
+ "grad_norm": 0.18146033585071564,
+ "learning_rate": 4.224530831099196e-06,
+ "loss": 0.5084807395935058,
+ "step": 5360
+ },
+ {
+ "epoch": 1.4424037269162449,
+ "grad_norm": 0.2251797467470169,
+ "learning_rate": 4.184316353887399e-06,
+ "loss": 0.5146170139312745,
+ "step": 5380
+ },
+ {
+ "epoch": 1.4477661963334114,
+ "grad_norm": 0.18744796514511108,
+ "learning_rate": 4.144101876675603e-06,
+ "loss": 0.5189927577972412,
+ "step": 5400
+ },
+ {
+ "epoch": 1.4531286657505782,
+ "grad_norm": 0.25737133622169495,
+ "learning_rate": 4.103887399463807e-06,
+ "loss": 0.4891658782958984,
+ "step": 5420
+ },
+ {
+ "epoch": 1.4584911351677448,
+ "grad_norm": 0.20580479502677917,
+ "learning_rate": 4.0636729222520105e-06,
+ "loss": 0.4953591823577881,
+ "step": 5440
+ },
+ {
+ "epoch": 1.4638536045849113,
+ "grad_norm": 0.2351546287536621,
+ "learning_rate": 4.023458445040214e-06,
+ "loss": 0.5025320053100586,
+ "step": 5460
+ },
+ {
+ "epoch": 1.4692160740020779,
+ "grad_norm": 0.1819481998682022,
+ "learning_rate": 3.983243967828418e-06,
+ "loss": 0.47151756286621094,
+ "step": 5480
+ },
+ {
+ "epoch": 1.4745785434192444,
+ "grad_norm": 0.20772472023963928,
+ "learning_rate": 3.943029490616622e-06,
+ "loss": 0.4678915023803711,
+ "step": 5500
+ },
+ {
+ "epoch": 1.4799410128364112,
+ "grad_norm": 0.2203037440776825,
+ "learning_rate": 3.902815013404825e-06,
+ "loss": 0.46007452011108396,
+ "step": 5520
+ },
+ {
+ "epoch": 1.4853034822535778,
+ "grad_norm": 0.15371400117874146,
+ "learning_rate": 3.86260053619303e-06,
+ "loss": 0.44407024383544924,
+ "step": 5540
+ },
+ {
+ "epoch": 1.4906659516707443,
+ "grad_norm": 0.2276080846786499,
+ "learning_rate": 3.8223860589812335e-06,
+ "loss": 0.4730556488037109,
+ "step": 5560
+ },
+ {
+ "epoch": 1.4960284210879111,
+ "grad_norm": 0.24482466280460358,
+ "learning_rate": 3.7821715817694376e-06,
+ "loss": 0.5073911666870117,
+ "step": 5580
+ },
+ {
+ "epoch": 1.5013908905050775,
+ "grad_norm": 0.20438458025455475,
+ "learning_rate": 3.741957104557641e-06,
+ "loss": 0.46701641082763673,
+ "step": 5600
+ },
+ {
+ "epoch": 1.5067533599222442,
+ "grad_norm": 0.19854313135147095,
+ "learning_rate": 3.7017426273458446e-06,
+ "loss": 0.46309399604797363,
+ "step": 5620
+ },
+ {
+ "epoch": 1.5121158293394108,
+ "grad_norm": 0.18356069922447205,
+ "learning_rate": 3.6615281501340483e-06,
+ "loss": 0.503613805770874,
+ "step": 5640
+ },
+ {
+ "epoch": 1.5174782987565774,
+ "grad_norm": 0.2009744495153427,
+ "learning_rate": 3.621313672922252e-06,
+ "loss": 0.4765054225921631,
+ "step": 5660
+ },
+ {
+ "epoch": 1.5228407681737441,
+ "grad_norm": 0.3058745563030243,
+ "learning_rate": 3.5810991957104557e-06,
+ "loss": 0.5179148197174073,
+ "step": 5680
+ },
+ {
+ "epoch": 1.5282032375909105,
+ "grad_norm": 0.17671597003936768,
+ "learning_rate": 3.54088471849866e-06,
+ "loss": 0.45907344818115237,
+ "step": 5700
+ },
+ {
+ "epoch": 1.5335657070080773,
+ "grad_norm": 0.22209160029888153,
+ "learning_rate": 3.5006702412868635e-06,
+ "loss": 0.49304862022399903,
+ "step": 5720
+ },
+ {
+ "epoch": 1.5389281764252438,
+ "grad_norm": 0.21018914878368378,
+ "learning_rate": 3.4604557640750672e-06,
+ "loss": 0.5536758422851562,
+ "step": 5740
+ },
+ {
+ "epoch": 1.5442906458424104,
+ "grad_norm": 0.14339996874332428,
+ "learning_rate": 3.420241286863271e-06,
+ "loss": 0.48726091384887693,
+ "step": 5760
+ },
+ {
+ "epoch": 1.5496531152595772,
+ "grad_norm": 0.11419746279716492,
+ "learning_rate": 3.3800268096514746e-06,
+ "loss": 0.4514151573181152,
+ "step": 5780
+ },
+ {
+ "epoch": 1.5550155846767435,
+ "grad_norm": 0.18168962001800537,
+ "learning_rate": 3.3398123324396783e-06,
+ "loss": 0.5279990196228027,
+ "step": 5800
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 7.141796059221197e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-5800/training_args.bin b/checkpoint-5800/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-5800/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-600/README.md b/checkpoint-600/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-600/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-600/adapter_config.json b/checkpoint-600/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-600/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-600/adapter_model.safetensors b/checkpoint-600/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..1b0beae437d9f3b7de85a27efcbbe36851e34a54
--- /dev/null
+++ b/checkpoint-600/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f1169af160e019a7373844aea7c18a0972e25fcc00f8db1ca2d7b37c1fc53cfa
+size 50360752
diff --git a/checkpoint-600/chat_template.jinja b/checkpoint-600/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-600/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-600/optimizer.pt b/checkpoint-600/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..41ac24db16c954b7c3a4f6b6d9025ddb075e428d
--- /dev/null
+++ b/checkpoint-600/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0f8e9ba6d8858deca4693da4556589ab009b0c0acd0f5709f9eea9cf3ddc048e
+size 100828235
diff --git a/checkpoint-600/rng_state.pth b/checkpoint-600/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..9bb823fba828655ac661a72989f8259de457c6a3
--- /dev/null
+++ b/checkpoint-600/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c8f869532707a1216d0a788bfd4dc1d65138d6bd118120f551cacd491bbb96b4
+size 14645
diff --git a/checkpoint-600/scheduler.pt b/checkpoint-600/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..d83fbc73c447eba14394d91a02cba49bb3135f12
--- /dev/null
+++ b/checkpoint-600/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21c2421bb1edc243038582e86c94a26d67145168f5438f0ca7b23a942e7dfa1e
+size 1465
diff --git a/checkpoint-600/tokenizer.json b/checkpoint-600/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-600/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-600/tokenizer_config.json b/checkpoint-600/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-600/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-600/trainer_state.json b/checkpoint-600/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..8744fac2b14347b1e915819a2dc59f9f44938ff9
--- /dev/null
+++ b/checkpoint-600/trainer_state.json
@@ -0,0 +1,244 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.16087408251499816,
+ "eval_steps": 500,
+ "global_step": 600,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 7.35254579186688e+16,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-600/training_args.bin b/checkpoint-600/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-600/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-6000/README.md b/checkpoint-6000/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-6000/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-6000/adapter_config.json b/checkpoint-6000/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-6000/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-6000/adapter_model.safetensors b/checkpoint-6000/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..3e8405737453d00c7da7cfa33ffa3072b1f74072
--- /dev/null
+++ b/checkpoint-6000/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1d7da8f2e0bd7e4fbddb5eb821bd5ba03f35be2409a0e5bdd187c58c86db0b05
+size 50360752
diff --git a/checkpoint-6000/chat_template.jinja b/checkpoint-6000/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-6000/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-6000/optimizer.pt b/checkpoint-6000/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..4765cada5d928d31a3664daf96a0eba8dc39d857
--- /dev/null
+++ b/checkpoint-6000/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:86954006a24eeedce074114af943943910cd56e82b6aa5b8f1e08768cbd71242
+size 100828235
diff --git a/checkpoint-6000/rng_state.pth b/checkpoint-6000/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..1b02f4f798a6e4e3655dc8af821c53672ef6a9cf
--- /dev/null
+++ b/checkpoint-6000/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d761063cf2e98a79d7f1635899deea6cdb52acd8964f54fe373e307bbf2cbad1
+size 14645
diff --git a/checkpoint-6000/scheduler.pt b/checkpoint-6000/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..f1a9fd4e828798e91c05aa7abf28c77d75b8bc33
--- /dev/null
+++ b/checkpoint-6000/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:6eb48adcee1fa0c3d7fb5270844c86c911e478414ee44487aa4d5df5179e49f9
+size 1465
diff --git a/checkpoint-6000/tokenizer.json b/checkpoint-6000/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-6000/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-6000/tokenizer_config.json b/checkpoint-6000/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-6000/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-6000/trainer_state.json b/checkpoint-6000/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..daa9046fb9562217c3a0cbd6a52feb9c315f611c
--- /dev/null
+++ b/checkpoint-6000/trainer_state.json
@@ -0,0 +1,2134 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.6086402788484095,
+ "eval_steps": 500,
+ "global_step": 6000,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ },
+ {
+ "epoch": 1.185005194892248,
+ "grad_norm": 0.16667844355106354,
+ "learning_rate": 6.114611260053619e-06,
+ "loss": 0.5159809589385986,
+ "step": 4420
+ },
+ {
+ "epoch": 1.1903676643094145,
+ "grad_norm": 0.19206570088863373,
+ "learning_rate": 6.074396782841823e-06,
+ "loss": 0.47541089057922364,
+ "step": 4440
+ },
+ {
+ "epoch": 1.195730133726581,
+ "grad_norm": 0.17394617199897766,
+ "learning_rate": 6.034182305630027e-06,
+ "loss": 0.5470661640167236,
+ "step": 4460
+ },
+ {
+ "epoch": 1.2010926031437477,
+ "grad_norm": 0.210404634475708,
+ "learning_rate": 5.993967828418231e-06,
+ "loss": 0.5377882957458496,
+ "step": 4480
+ },
+ {
+ "epoch": 1.2064550725609142,
+ "grad_norm": 0.18084648251533508,
+ "learning_rate": 5.953753351206435e-06,
+ "loss": 0.5037185192108155,
+ "step": 4500
+ },
+ {
+ "epoch": 1.211817541978081,
+ "grad_norm": 0.23707027733325958,
+ "learning_rate": 5.913538873994639e-06,
+ "loss": 0.4822190284729004,
+ "step": 4520
+ },
+ {
+ "epoch": 1.2171800113952476,
+ "grad_norm": 0.16474473476409912,
+ "learning_rate": 5.873324396782842e-06,
+ "loss": 0.46645288467407225,
+ "step": 4540
+ },
+ {
+ "epoch": 1.2225424808124141,
+ "grad_norm": 0.2142348438501358,
+ "learning_rate": 5.833109919571046e-06,
+ "loss": 0.5255855560302735,
+ "step": 4560
+ },
+ {
+ "epoch": 1.2279049502295807,
+ "grad_norm": 0.2531765103340149,
+ "learning_rate": 5.79289544235925e-06,
+ "loss": 0.507044792175293,
+ "step": 4580
+ },
+ {
+ "epoch": 1.2332674196467472,
+ "grad_norm": 0.2553550899028778,
+ "learning_rate": 5.7526809651474535e-06,
+ "loss": 0.4767824649810791,
+ "step": 4600
+ },
+ {
+ "epoch": 1.238629889063914,
+ "grad_norm": 0.14484412968158722,
+ "learning_rate": 5.712466487935657e-06,
+ "loss": 0.4675601005554199,
+ "step": 4620
+ },
+ {
+ "epoch": 1.2439923584810806,
+ "grad_norm": 0.14328251779079437,
+ "learning_rate": 5.672252010723861e-06,
+ "loss": 0.4956005573272705,
+ "step": 4640
+ },
+ {
+ "epoch": 1.2493548278982471,
+ "grad_norm": 0.1739245355129242,
+ "learning_rate": 5.632037533512065e-06,
+ "loss": 0.48583345413208007,
+ "step": 4660
+ },
+ {
+ "epoch": 1.2547172973154137,
+ "grad_norm": 0.21294184029102325,
+ "learning_rate": 5.591823056300268e-06,
+ "loss": 0.520921277999878,
+ "step": 4680
+ },
+ {
+ "epoch": 1.2600797667325803,
+ "grad_norm": 0.25132355093955994,
+ "learning_rate": 5.551608579088472e-06,
+ "loss": 0.5295385837554931,
+ "step": 4700
+ },
+ {
+ "epoch": 1.265442236149747,
+ "grad_norm": 0.18603841960430145,
+ "learning_rate": 5.511394101876676e-06,
+ "loss": 0.47570199966430665,
+ "step": 4720
+ },
+ {
+ "epoch": 1.2708047055669136,
+ "grad_norm": 0.19883134961128235,
+ "learning_rate": 5.471179624664879e-06,
+ "loss": 0.5016080379486084,
+ "step": 4740
+ },
+ {
+ "epoch": 1.2761671749840802,
+ "grad_norm": 0.19640181958675385,
+ "learning_rate": 5.430965147453083e-06,
+ "loss": 0.4999081134796143,
+ "step": 4760
+ },
+ {
+ "epoch": 1.2815296444012467,
+ "grad_norm": 0.2584764361381531,
+ "learning_rate": 5.390750670241287e-06,
+ "loss": 0.4780082702636719,
+ "step": 4780
+ },
+ {
+ "epoch": 1.2868921138184133,
+ "grad_norm": 0.2925741374492645,
+ "learning_rate": 5.3505361930294905e-06,
+ "loss": 0.5131395816802978,
+ "step": 4800
+ },
+ {
+ "epoch": 1.29225458323558,
+ "grad_norm": 0.18971531093120575,
+ "learning_rate": 5.310321715817694e-06,
+ "loss": 0.455674409866333,
+ "step": 4820
+ },
+ {
+ "epoch": 1.2976170526527466,
+ "grad_norm": 0.16778405010700226,
+ "learning_rate": 5.270107238605898e-06,
+ "loss": 0.5070962905883789,
+ "step": 4840
+ },
+ {
+ "epoch": 1.3029795220699132,
+ "grad_norm": 0.30026957392692566,
+ "learning_rate": 5.2298927613941016e-06,
+ "loss": 0.5120027542114258,
+ "step": 4860
+ },
+ {
+ "epoch": 1.3083419914870797,
+ "grad_norm": 0.17846634984016418,
+ "learning_rate": 5.189678284182305e-06,
+ "loss": 0.5114477157592774,
+ "step": 4880
+ },
+ {
+ "epoch": 1.3137044609042463,
+ "grad_norm": 0.1962418258190155,
+ "learning_rate": 5.149463806970509e-06,
+ "loss": 0.5043613910675049,
+ "step": 4900
+ },
+ {
+ "epoch": 1.319066930321413,
+ "grad_norm": 0.18446756899356842,
+ "learning_rate": 5.1092493297587135e-06,
+ "loss": 0.5396455287933349,
+ "step": 4920
+ },
+ {
+ "epoch": 1.3244293997385796,
+ "grad_norm": 0.20886844396591187,
+ "learning_rate": 5.069034852546917e-06,
+ "loss": 0.4879767417907715,
+ "step": 4940
+ },
+ {
+ "epoch": 1.3297918691557462,
+ "grad_norm": 0.16687901318073273,
+ "learning_rate": 5.028820375335121e-06,
+ "loss": 0.5014327049255372,
+ "step": 4960
+ },
+ {
+ "epoch": 1.3351543385729128,
+ "grad_norm": 0.19595153629779816,
+ "learning_rate": 4.988605898123325e-06,
+ "loss": 0.5375277996063232,
+ "step": 4980
+ },
+ {
+ "epoch": 1.3405168079900793,
+ "grad_norm": 0.2372344732284546,
+ "learning_rate": 4.948391420911528e-06,
+ "loss": 0.5020076274871826,
+ "step": 5000
+ },
+ {
+ "epoch": 1.345879277407246,
+ "grad_norm": 0.21030014753341675,
+ "learning_rate": 4.908176943699732e-06,
+ "loss": 0.5111066818237304,
+ "step": 5020
+ },
+ {
+ "epoch": 1.3512417468244127,
+ "grad_norm": 0.1866692751646042,
+ "learning_rate": 4.867962466487936e-06,
+ "loss": 0.4515383720397949,
+ "step": 5040
+ },
+ {
+ "epoch": 1.3566042162415792,
+ "grad_norm": 0.22531798481941223,
+ "learning_rate": 4.827747989276139e-06,
+ "loss": 0.4757690906524658,
+ "step": 5060
+ },
+ {
+ "epoch": 1.3619666856587458,
+ "grad_norm": 0.15868768095970154,
+ "learning_rate": 4.787533512064343e-06,
+ "loss": 0.45842318534851073,
+ "step": 5080
+ },
+ {
+ "epoch": 1.3673291550759124,
+ "grad_norm": 0.24528546631336212,
+ "learning_rate": 4.747319034852547e-06,
+ "loss": 0.47269258499145506,
+ "step": 5100
+ },
+ {
+ "epoch": 1.3726916244930791,
+ "grad_norm": 0.17387732863426208,
+ "learning_rate": 4.707104557640751e-06,
+ "loss": 0.5103805065155029,
+ "step": 5120
+ },
+ {
+ "epoch": 1.3780540939102457,
+ "grad_norm": 0.20686905086040497,
+ "learning_rate": 4.666890080428955e-06,
+ "loss": 0.5135180950164795,
+ "step": 5140
+ },
+ {
+ "epoch": 1.3834165633274123,
+ "grad_norm": 0.19599783420562744,
+ "learning_rate": 4.626675603217159e-06,
+ "loss": 0.5045839786529541,
+ "step": 5160
+ },
+ {
+ "epoch": 1.3887790327445788,
+ "grad_norm": 0.2585010528564453,
+ "learning_rate": 4.586461126005362e-06,
+ "loss": 0.45903496742248534,
+ "step": 5180
+ },
+ {
+ "epoch": 1.3941415021617454,
+ "grad_norm": 0.1688319593667984,
+ "learning_rate": 4.546246648793566e-06,
+ "loss": 0.5017509937286377,
+ "step": 5200
+ },
+ {
+ "epoch": 1.3995039715789122,
+ "grad_norm": 0.21520815789699554,
+ "learning_rate": 4.50603217158177e-06,
+ "loss": 0.48459539413452146,
+ "step": 5220
+ },
+ {
+ "epoch": 1.4048664409960787,
+ "grad_norm": 0.20514647662639618,
+ "learning_rate": 4.4658176943699735e-06,
+ "loss": 0.5073423862457276,
+ "step": 5240
+ },
+ {
+ "epoch": 1.4102289104132453,
+ "grad_norm": 0.21835413575172424,
+ "learning_rate": 4.425603217158177e-06,
+ "loss": 0.5290310382843018,
+ "step": 5260
+ },
+ {
+ "epoch": 1.4155913798304118,
+ "grad_norm": 0.28042587637901306,
+ "learning_rate": 4.385388739946381e-06,
+ "loss": 0.4823312759399414,
+ "step": 5280
+ },
+ {
+ "epoch": 1.4209538492475784,
+ "grad_norm": 0.18959026038646698,
+ "learning_rate": 4.345174262734585e-06,
+ "loss": 0.4921241760253906,
+ "step": 5300
+ },
+ {
+ "epoch": 1.4263163186647452,
+ "grad_norm": 0.18584316968917847,
+ "learning_rate": 4.304959785522788e-06,
+ "loss": 0.4892130374908447,
+ "step": 5320
+ },
+ {
+ "epoch": 1.4316787880819117,
+ "grad_norm": 0.17588038742542267,
+ "learning_rate": 4.264745308310992e-06,
+ "loss": 0.4822041988372803,
+ "step": 5340
+ },
+ {
+ "epoch": 1.4370412574990783,
+ "grad_norm": 0.18146033585071564,
+ "learning_rate": 4.224530831099196e-06,
+ "loss": 0.5084807395935058,
+ "step": 5360
+ },
+ {
+ "epoch": 1.4424037269162449,
+ "grad_norm": 0.2251797467470169,
+ "learning_rate": 4.184316353887399e-06,
+ "loss": 0.5146170139312745,
+ "step": 5380
+ },
+ {
+ "epoch": 1.4477661963334114,
+ "grad_norm": 0.18744796514511108,
+ "learning_rate": 4.144101876675603e-06,
+ "loss": 0.5189927577972412,
+ "step": 5400
+ },
+ {
+ "epoch": 1.4531286657505782,
+ "grad_norm": 0.25737133622169495,
+ "learning_rate": 4.103887399463807e-06,
+ "loss": 0.4891658782958984,
+ "step": 5420
+ },
+ {
+ "epoch": 1.4584911351677448,
+ "grad_norm": 0.20580479502677917,
+ "learning_rate": 4.0636729222520105e-06,
+ "loss": 0.4953591823577881,
+ "step": 5440
+ },
+ {
+ "epoch": 1.4638536045849113,
+ "grad_norm": 0.2351546287536621,
+ "learning_rate": 4.023458445040214e-06,
+ "loss": 0.5025320053100586,
+ "step": 5460
+ },
+ {
+ "epoch": 1.4692160740020779,
+ "grad_norm": 0.1819481998682022,
+ "learning_rate": 3.983243967828418e-06,
+ "loss": 0.47151756286621094,
+ "step": 5480
+ },
+ {
+ "epoch": 1.4745785434192444,
+ "grad_norm": 0.20772472023963928,
+ "learning_rate": 3.943029490616622e-06,
+ "loss": 0.4678915023803711,
+ "step": 5500
+ },
+ {
+ "epoch": 1.4799410128364112,
+ "grad_norm": 0.2203037440776825,
+ "learning_rate": 3.902815013404825e-06,
+ "loss": 0.46007452011108396,
+ "step": 5520
+ },
+ {
+ "epoch": 1.4853034822535778,
+ "grad_norm": 0.15371400117874146,
+ "learning_rate": 3.86260053619303e-06,
+ "loss": 0.44407024383544924,
+ "step": 5540
+ },
+ {
+ "epoch": 1.4906659516707443,
+ "grad_norm": 0.2276080846786499,
+ "learning_rate": 3.8223860589812335e-06,
+ "loss": 0.4730556488037109,
+ "step": 5560
+ },
+ {
+ "epoch": 1.4960284210879111,
+ "grad_norm": 0.24482466280460358,
+ "learning_rate": 3.7821715817694376e-06,
+ "loss": 0.5073911666870117,
+ "step": 5580
+ },
+ {
+ "epoch": 1.5013908905050775,
+ "grad_norm": 0.20438458025455475,
+ "learning_rate": 3.741957104557641e-06,
+ "loss": 0.46701641082763673,
+ "step": 5600
+ },
+ {
+ "epoch": 1.5067533599222442,
+ "grad_norm": 0.19854313135147095,
+ "learning_rate": 3.7017426273458446e-06,
+ "loss": 0.46309399604797363,
+ "step": 5620
+ },
+ {
+ "epoch": 1.5121158293394108,
+ "grad_norm": 0.18356069922447205,
+ "learning_rate": 3.6615281501340483e-06,
+ "loss": 0.503613805770874,
+ "step": 5640
+ },
+ {
+ "epoch": 1.5174782987565774,
+ "grad_norm": 0.2009744495153427,
+ "learning_rate": 3.621313672922252e-06,
+ "loss": 0.4765054225921631,
+ "step": 5660
+ },
+ {
+ "epoch": 1.5228407681737441,
+ "grad_norm": 0.3058745563030243,
+ "learning_rate": 3.5810991957104557e-06,
+ "loss": 0.5179148197174073,
+ "step": 5680
+ },
+ {
+ "epoch": 1.5282032375909105,
+ "grad_norm": 0.17671597003936768,
+ "learning_rate": 3.54088471849866e-06,
+ "loss": 0.45907344818115237,
+ "step": 5700
+ },
+ {
+ "epoch": 1.5335657070080773,
+ "grad_norm": 0.22209160029888153,
+ "learning_rate": 3.5006702412868635e-06,
+ "loss": 0.49304862022399903,
+ "step": 5720
+ },
+ {
+ "epoch": 1.5389281764252438,
+ "grad_norm": 0.21018914878368378,
+ "learning_rate": 3.4604557640750672e-06,
+ "loss": 0.5536758422851562,
+ "step": 5740
+ },
+ {
+ "epoch": 1.5442906458424104,
+ "grad_norm": 0.14339996874332428,
+ "learning_rate": 3.420241286863271e-06,
+ "loss": 0.48726091384887693,
+ "step": 5760
+ },
+ {
+ "epoch": 1.5496531152595772,
+ "grad_norm": 0.11419746279716492,
+ "learning_rate": 3.3800268096514746e-06,
+ "loss": 0.4514151573181152,
+ "step": 5780
+ },
+ {
+ "epoch": 1.5550155846767435,
+ "grad_norm": 0.18168962001800537,
+ "learning_rate": 3.3398123324396783e-06,
+ "loss": 0.5279990196228027,
+ "step": 5800
+ },
+ {
+ "epoch": 1.5603780540939103,
+ "grad_norm": 0.24244488775730133,
+ "learning_rate": 3.299597855227882e-06,
+ "loss": 0.49297361373901366,
+ "step": 5820
+ },
+ {
+ "epoch": 1.5657405235110768,
+ "grad_norm": 0.2017296999692917,
+ "learning_rate": 3.2593833780160857e-06,
+ "loss": 0.49305019378662107,
+ "step": 5840
+ },
+ {
+ "epoch": 1.5711029929282434,
+ "grad_norm": 0.22592377662658691,
+ "learning_rate": 3.2191689008042894e-06,
+ "loss": 0.4862989902496338,
+ "step": 5860
+ },
+ {
+ "epoch": 1.5764654623454102,
+ "grad_norm": 0.24772357940673828,
+ "learning_rate": 3.1789544235924935e-06,
+ "loss": 0.45182647705078127,
+ "step": 5880
+ },
+ {
+ "epoch": 1.5818279317625765,
+ "grad_norm": 0.20607218146324158,
+ "learning_rate": 3.1387399463806972e-06,
+ "loss": 0.48905248641967775,
+ "step": 5900
+ },
+ {
+ "epoch": 1.5871904011797433,
+ "grad_norm": 0.1931353509426117,
+ "learning_rate": 3.098525469168901e-06,
+ "loss": 0.5307461261749268,
+ "step": 5920
+ },
+ {
+ "epoch": 1.5925528705969099,
+ "grad_norm": 0.16020581126213074,
+ "learning_rate": 3.0583109919571046e-06,
+ "loss": 0.4672811985015869,
+ "step": 5940
+ },
+ {
+ "epoch": 1.5979153400140764,
+ "grad_norm": 0.23668015003204346,
+ "learning_rate": 3.0180965147453083e-06,
+ "loss": 0.5272688865661621,
+ "step": 5960
+ },
+ {
+ "epoch": 1.6032778094312432,
+ "grad_norm": 0.1916576772928238,
+ "learning_rate": 2.977882037533512e-06,
+ "loss": 0.4859332084655762,
+ "step": 5980
+ },
+ {
+ "epoch": 1.6086402788484095,
+ "grad_norm": 0.23635101318359375,
+ "learning_rate": 2.9376675603217157e-06,
+ "loss": 0.5418910980224609,
+ "step": 6000
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 7.388858570735186e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-6000/training_args.bin b/checkpoint-6000/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-6000/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-6200/README.md b/checkpoint-6200/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-6200/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-6200/adapter_config.json b/checkpoint-6200/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-6200/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-6200/adapter_model.safetensors b/checkpoint-6200/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..67f488e84b33f07e28f288afaea202eda2de8ff6
--- /dev/null
+++ b/checkpoint-6200/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ecddfa1ad15326aa3aad668ace2624b7406005818d1ccc033c855dd65daa0af1
+size 50360752
diff --git a/checkpoint-6200/chat_template.jinja b/checkpoint-6200/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-6200/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-6200/optimizer.pt b/checkpoint-6200/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..3b6371526b888677e438cb71d951d9003aee3ec6
--- /dev/null
+++ b/checkpoint-6200/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:063c20a9fc6d1290e39711ff8109126773c0d1e8e51e63de2ba6d87f78f40c1f
+size 100828235
diff --git a/checkpoint-6200/rng_state.pth b/checkpoint-6200/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..181e79122595a400a196a12bc39acf21da26f8ba
--- /dev/null
+++ b/checkpoint-6200/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:330ee16477886512f1c2dcb9d2dd56ad6f274db1cf0feb2197d89de7803d2c2a
+size 14645
diff --git a/checkpoint-6200/scheduler.pt b/checkpoint-6200/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..db9e284ba6b2e4d27e044d4fff2c2a7b3182e6da
--- /dev/null
+++ b/checkpoint-6200/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:9a33bf06e2136cdac84aaeda236d77d618aa9aabbf0388ae72b32a51bfebf8fa
+size 1465
diff --git a/checkpoint-6200/tokenizer.json b/checkpoint-6200/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-6200/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-6200/tokenizer_config.json b/checkpoint-6200/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-6200/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-6200/trainer_state.json b/checkpoint-6200/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..87d941ff821af8c115342b086348d4cd693a43ee
--- /dev/null
+++ b/checkpoint-6200/trainer_state.json
@@ -0,0 +1,2204 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.6622649730200756,
+ "eval_steps": 500,
+ "global_step": 6200,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ },
+ {
+ "epoch": 1.185005194892248,
+ "grad_norm": 0.16667844355106354,
+ "learning_rate": 6.114611260053619e-06,
+ "loss": 0.5159809589385986,
+ "step": 4420
+ },
+ {
+ "epoch": 1.1903676643094145,
+ "grad_norm": 0.19206570088863373,
+ "learning_rate": 6.074396782841823e-06,
+ "loss": 0.47541089057922364,
+ "step": 4440
+ },
+ {
+ "epoch": 1.195730133726581,
+ "grad_norm": 0.17394617199897766,
+ "learning_rate": 6.034182305630027e-06,
+ "loss": 0.5470661640167236,
+ "step": 4460
+ },
+ {
+ "epoch": 1.2010926031437477,
+ "grad_norm": 0.210404634475708,
+ "learning_rate": 5.993967828418231e-06,
+ "loss": 0.5377882957458496,
+ "step": 4480
+ },
+ {
+ "epoch": 1.2064550725609142,
+ "grad_norm": 0.18084648251533508,
+ "learning_rate": 5.953753351206435e-06,
+ "loss": 0.5037185192108155,
+ "step": 4500
+ },
+ {
+ "epoch": 1.211817541978081,
+ "grad_norm": 0.23707027733325958,
+ "learning_rate": 5.913538873994639e-06,
+ "loss": 0.4822190284729004,
+ "step": 4520
+ },
+ {
+ "epoch": 1.2171800113952476,
+ "grad_norm": 0.16474473476409912,
+ "learning_rate": 5.873324396782842e-06,
+ "loss": 0.46645288467407225,
+ "step": 4540
+ },
+ {
+ "epoch": 1.2225424808124141,
+ "grad_norm": 0.2142348438501358,
+ "learning_rate": 5.833109919571046e-06,
+ "loss": 0.5255855560302735,
+ "step": 4560
+ },
+ {
+ "epoch": 1.2279049502295807,
+ "grad_norm": 0.2531765103340149,
+ "learning_rate": 5.79289544235925e-06,
+ "loss": 0.507044792175293,
+ "step": 4580
+ },
+ {
+ "epoch": 1.2332674196467472,
+ "grad_norm": 0.2553550899028778,
+ "learning_rate": 5.7526809651474535e-06,
+ "loss": 0.4767824649810791,
+ "step": 4600
+ },
+ {
+ "epoch": 1.238629889063914,
+ "grad_norm": 0.14484412968158722,
+ "learning_rate": 5.712466487935657e-06,
+ "loss": 0.4675601005554199,
+ "step": 4620
+ },
+ {
+ "epoch": 1.2439923584810806,
+ "grad_norm": 0.14328251779079437,
+ "learning_rate": 5.672252010723861e-06,
+ "loss": 0.4956005573272705,
+ "step": 4640
+ },
+ {
+ "epoch": 1.2493548278982471,
+ "grad_norm": 0.1739245355129242,
+ "learning_rate": 5.632037533512065e-06,
+ "loss": 0.48583345413208007,
+ "step": 4660
+ },
+ {
+ "epoch": 1.2547172973154137,
+ "grad_norm": 0.21294184029102325,
+ "learning_rate": 5.591823056300268e-06,
+ "loss": 0.520921277999878,
+ "step": 4680
+ },
+ {
+ "epoch": 1.2600797667325803,
+ "grad_norm": 0.25132355093955994,
+ "learning_rate": 5.551608579088472e-06,
+ "loss": 0.5295385837554931,
+ "step": 4700
+ },
+ {
+ "epoch": 1.265442236149747,
+ "grad_norm": 0.18603841960430145,
+ "learning_rate": 5.511394101876676e-06,
+ "loss": 0.47570199966430665,
+ "step": 4720
+ },
+ {
+ "epoch": 1.2708047055669136,
+ "grad_norm": 0.19883134961128235,
+ "learning_rate": 5.471179624664879e-06,
+ "loss": 0.5016080379486084,
+ "step": 4740
+ },
+ {
+ "epoch": 1.2761671749840802,
+ "grad_norm": 0.19640181958675385,
+ "learning_rate": 5.430965147453083e-06,
+ "loss": 0.4999081134796143,
+ "step": 4760
+ },
+ {
+ "epoch": 1.2815296444012467,
+ "grad_norm": 0.2584764361381531,
+ "learning_rate": 5.390750670241287e-06,
+ "loss": 0.4780082702636719,
+ "step": 4780
+ },
+ {
+ "epoch": 1.2868921138184133,
+ "grad_norm": 0.2925741374492645,
+ "learning_rate": 5.3505361930294905e-06,
+ "loss": 0.5131395816802978,
+ "step": 4800
+ },
+ {
+ "epoch": 1.29225458323558,
+ "grad_norm": 0.18971531093120575,
+ "learning_rate": 5.310321715817694e-06,
+ "loss": 0.455674409866333,
+ "step": 4820
+ },
+ {
+ "epoch": 1.2976170526527466,
+ "grad_norm": 0.16778405010700226,
+ "learning_rate": 5.270107238605898e-06,
+ "loss": 0.5070962905883789,
+ "step": 4840
+ },
+ {
+ "epoch": 1.3029795220699132,
+ "grad_norm": 0.30026957392692566,
+ "learning_rate": 5.2298927613941016e-06,
+ "loss": 0.5120027542114258,
+ "step": 4860
+ },
+ {
+ "epoch": 1.3083419914870797,
+ "grad_norm": 0.17846634984016418,
+ "learning_rate": 5.189678284182305e-06,
+ "loss": 0.5114477157592774,
+ "step": 4880
+ },
+ {
+ "epoch": 1.3137044609042463,
+ "grad_norm": 0.1962418258190155,
+ "learning_rate": 5.149463806970509e-06,
+ "loss": 0.5043613910675049,
+ "step": 4900
+ },
+ {
+ "epoch": 1.319066930321413,
+ "grad_norm": 0.18446756899356842,
+ "learning_rate": 5.1092493297587135e-06,
+ "loss": 0.5396455287933349,
+ "step": 4920
+ },
+ {
+ "epoch": 1.3244293997385796,
+ "grad_norm": 0.20886844396591187,
+ "learning_rate": 5.069034852546917e-06,
+ "loss": 0.4879767417907715,
+ "step": 4940
+ },
+ {
+ "epoch": 1.3297918691557462,
+ "grad_norm": 0.16687901318073273,
+ "learning_rate": 5.028820375335121e-06,
+ "loss": 0.5014327049255372,
+ "step": 4960
+ },
+ {
+ "epoch": 1.3351543385729128,
+ "grad_norm": 0.19595153629779816,
+ "learning_rate": 4.988605898123325e-06,
+ "loss": 0.5375277996063232,
+ "step": 4980
+ },
+ {
+ "epoch": 1.3405168079900793,
+ "grad_norm": 0.2372344732284546,
+ "learning_rate": 4.948391420911528e-06,
+ "loss": 0.5020076274871826,
+ "step": 5000
+ },
+ {
+ "epoch": 1.345879277407246,
+ "grad_norm": 0.21030014753341675,
+ "learning_rate": 4.908176943699732e-06,
+ "loss": 0.5111066818237304,
+ "step": 5020
+ },
+ {
+ "epoch": 1.3512417468244127,
+ "grad_norm": 0.1866692751646042,
+ "learning_rate": 4.867962466487936e-06,
+ "loss": 0.4515383720397949,
+ "step": 5040
+ },
+ {
+ "epoch": 1.3566042162415792,
+ "grad_norm": 0.22531798481941223,
+ "learning_rate": 4.827747989276139e-06,
+ "loss": 0.4757690906524658,
+ "step": 5060
+ },
+ {
+ "epoch": 1.3619666856587458,
+ "grad_norm": 0.15868768095970154,
+ "learning_rate": 4.787533512064343e-06,
+ "loss": 0.45842318534851073,
+ "step": 5080
+ },
+ {
+ "epoch": 1.3673291550759124,
+ "grad_norm": 0.24528546631336212,
+ "learning_rate": 4.747319034852547e-06,
+ "loss": 0.47269258499145506,
+ "step": 5100
+ },
+ {
+ "epoch": 1.3726916244930791,
+ "grad_norm": 0.17387732863426208,
+ "learning_rate": 4.707104557640751e-06,
+ "loss": 0.5103805065155029,
+ "step": 5120
+ },
+ {
+ "epoch": 1.3780540939102457,
+ "grad_norm": 0.20686905086040497,
+ "learning_rate": 4.666890080428955e-06,
+ "loss": 0.5135180950164795,
+ "step": 5140
+ },
+ {
+ "epoch": 1.3834165633274123,
+ "grad_norm": 0.19599783420562744,
+ "learning_rate": 4.626675603217159e-06,
+ "loss": 0.5045839786529541,
+ "step": 5160
+ },
+ {
+ "epoch": 1.3887790327445788,
+ "grad_norm": 0.2585010528564453,
+ "learning_rate": 4.586461126005362e-06,
+ "loss": 0.45903496742248534,
+ "step": 5180
+ },
+ {
+ "epoch": 1.3941415021617454,
+ "grad_norm": 0.1688319593667984,
+ "learning_rate": 4.546246648793566e-06,
+ "loss": 0.5017509937286377,
+ "step": 5200
+ },
+ {
+ "epoch": 1.3995039715789122,
+ "grad_norm": 0.21520815789699554,
+ "learning_rate": 4.50603217158177e-06,
+ "loss": 0.48459539413452146,
+ "step": 5220
+ },
+ {
+ "epoch": 1.4048664409960787,
+ "grad_norm": 0.20514647662639618,
+ "learning_rate": 4.4658176943699735e-06,
+ "loss": 0.5073423862457276,
+ "step": 5240
+ },
+ {
+ "epoch": 1.4102289104132453,
+ "grad_norm": 0.21835413575172424,
+ "learning_rate": 4.425603217158177e-06,
+ "loss": 0.5290310382843018,
+ "step": 5260
+ },
+ {
+ "epoch": 1.4155913798304118,
+ "grad_norm": 0.28042587637901306,
+ "learning_rate": 4.385388739946381e-06,
+ "loss": 0.4823312759399414,
+ "step": 5280
+ },
+ {
+ "epoch": 1.4209538492475784,
+ "grad_norm": 0.18959026038646698,
+ "learning_rate": 4.345174262734585e-06,
+ "loss": 0.4921241760253906,
+ "step": 5300
+ },
+ {
+ "epoch": 1.4263163186647452,
+ "grad_norm": 0.18584316968917847,
+ "learning_rate": 4.304959785522788e-06,
+ "loss": 0.4892130374908447,
+ "step": 5320
+ },
+ {
+ "epoch": 1.4316787880819117,
+ "grad_norm": 0.17588038742542267,
+ "learning_rate": 4.264745308310992e-06,
+ "loss": 0.4822041988372803,
+ "step": 5340
+ },
+ {
+ "epoch": 1.4370412574990783,
+ "grad_norm": 0.18146033585071564,
+ "learning_rate": 4.224530831099196e-06,
+ "loss": 0.5084807395935058,
+ "step": 5360
+ },
+ {
+ "epoch": 1.4424037269162449,
+ "grad_norm": 0.2251797467470169,
+ "learning_rate": 4.184316353887399e-06,
+ "loss": 0.5146170139312745,
+ "step": 5380
+ },
+ {
+ "epoch": 1.4477661963334114,
+ "grad_norm": 0.18744796514511108,
+ "learning_rate": 4.144101876675603e-06,
+ "loss": 0.5189927577972412,
+ "step": 5400
+ },
+ {
+ "epoch": 1.4531286657505782,
+ "grad_norm": 0.25737133622169495,
+ "learning_rate": 4.103887399463807e-06,
+ "loss": 0.4891658782958984,
+ "step": 5420
+ },
+ {
+ "epoch": 1.4584911351677448,
+ "grad_norm": 0.20580479502677917,
+ "learning_rate": 4.0636729222520105e-06,
+ "loss": 0.4953591823577881,
+ "step": 5440
+ },
+ {
+ "epoch": 1.4638536045849113,
+ "grad_norm": 0.2351546287536621,
+ "learning_rate": 4.023458445040214e-06,
+ "loss": 0.5025320053100586,
+ "step": 5460
+ },
+ {
+ "epoch": 1.4692160740020779,
+ "grad_norm": 0.1819481998682022,
+ "learning_rate": 3.983243967828418e-06,
+ "loss": 0.47151756286621094,
+ "step": 5480
+ },
+ {
+ "epoch": 1.4745785434192444,
+ "grad_norm": 0.20772472023963928,
+ "learning_rate": 3.943029490616622e-06,
+ "loss": 0.4678915023803711,
+ "step": 5500
+ },
+ {
+ "epoch": 1.4799410128364112,
+ "grad_norm": 0.2203037440776825,
+ "learning_rate": 3.902815013404825e-06,
+ "loss": 0.46007452011108396,
+ "step": 5520
+ },
+ {
+ "epoch": 1.4853034822535778,
+ "grad_norm": 0.15371400117874146,
+ "learning_rate": 3.86260053619303e-06,
+ "loss": 0.44407024383544924,
+ "step": 5540
+ },
+ {
+ "epoch": 1.4906659516707443,
+ "grad_norm": 0.2276080846786499,
+ "learning_rate": 3.8223860589812335e-06,
+ "loss": 0.4730556488037109,
+ "step": 5560
+ },
+ {
+ "epoch": 1.4960284210879111,
+ "grad_norm": 0.24482466280460358,
+ "learning_rate": 3.7821715817694376e-06,
+ "loss": 0.5073911666870117,
+ "step": 5580
+ },
+ {
+ "epoch": 1.5013908905050775,
+ "grad_norm": 0.20438458025455475,
+ "learning_rate": 3.741957104557641e-06,
+ "loss": 0.46701641082763673,
+ "step": 5600
+ },
+ {
+ "epoch": 1.5067533599222442,
+ "grad_norm": 0.19854313135147095,
+ "learning_rate": 3.7017426273458446e-06,
+ "loss": 0.46309399604797363,
+ "step": 5620
+ },
+ {
+ "epoch": 1.5121158293394108,
+ "grad_norm": 0.18356069922447205,
+ "learning_rate": 3.6615281501340483e-06,
+ "loss": 0.503613805770874,
+ "step": 5640
+ },
+ {
+ "epoch": 1.5174782987565774,
+ "grad_norm": 0.2009744495153427,
+ "learning_rate": 3.621313672922252e-06,
+ "loss": 0.4765054225921631,
+ "step": 5660
+ },
+ {
+ "epoch": 1.5228407681737441,
+ "grad_norm": 0.3058745563030243,
+ "learning_rate": 3.5810991957104557e-06,
+ "loss": 0.5179148197174073,
+ "step": 5680
+ },
+ {
+ "epoch": 1.5282032375909105,
+ "grad_norm": 0.17671597003936768,
+ "learning_rate": 3.54088471849866e-06,
+ "loss": 0.45907344818115237,
+ "step": 5700
+ },
+ {
+ "epoch": 1.5335657070080773,
+ "grad_norm": 0.22209160029888153,
+ "learning_rate": 3.5006702412868635e-06,
+ "loss": 0.49304862022399903,
+ "step": 5720
+ },
+ {
+ "epoch": 1.5389281764252438,
+ "grad_norm": 0.21018914878368378,
+ "learning_rate": 3.4604557640750672e-06,
+ "loss": 0.5536758422851562,
+ "step": 5740
+ },
+ {
+ "epoch": 1.5442906458424104,
+ "grad_norm": 0.14339996874332428,
+ "learning_rate": 3.420241286863271e-06,
+ "loss": 0.48726091384887693,
+ "step": 5760
+ },
+ {
+ "epoch": 1.5496531152595772,
+ "grad_norm": 0.11419746279716492,
+ "learning_rate": 3.3800268096514746e-06,
+ "loss": 0.4514151573181152,
+ "step": 5780
+ },
+ {
+ "epoch": 1.5550155846767435,
+ "grad_norm": 0.18168962001800537,
+ "learning_rate": 3.3398123324396783e-06,
+ "loss": 0.5279990196228027,
+ "step": 5800
+ },
+ {
+ "epoch": 1.5603780540939103,
+ "grad_norm": 0.24244488775730133,
+ "learning_rate": 3.299597855227882e-06,
+ "loss": 0.49297361373901366,
+ "step": 5820
+ },
+ {
+ "epoch": 1.5657405235110768,
+ "grad_norm": 0.2017296999692917,
+ "learning_rate": 3.2593833780160857e-06,
+ "loss": 0.49305019378662107,
+ "step": 5840
+ },
+ {
+ "epoch": 1.5711029929282434,
+ "grad_norm": 0.22592377662658691,
+ "learning_rate": 3.2191689008042894e-06,
+ "loss": 0.4862989902496338,
+ "step": 5860
+ },
+ {
+ "epoch": 1.5764654623454102,
+ "grad_norm": 0.24772357940673828,
+ "learning_rate": 3.1789544235924935e-06,
+ "loss": 0.45182647705078127,
+ "step": 5880
+ },
+ {
+ "epoch": 1.5818279317625765,
+ "grad_norm": 0.20607218146324158,
+ "learning_rate": 3.1387399463806972e-06,
+ "loss": 0.48905248641967775,
+ "step": 5900
+ },
+ {
+ "epoch": 1.5871904011797433,
+ "grad_norm": 0.1931353509426117,
+ "learning_rate": 3.098525469168901e-06,
+ "loss": 0.5307461261749268,
+ "step": 5920
+ },
+ {
+ "epoch": 1.5925528705969099,
+ "grad_norm": 0.16020581126213074,
+ "learning_rate": 3.0583109919571046e-06,
+ "loss": 0.4672811985015869,
+ "step": 5940
+ },
+ {
+ "epoch": 1.5979153400140764,
+ "grad_norm": 0.23668015003204346,
+ "learning_rate": 3.0180965147453083e-06,
+ "loss": 0.5272688865661621,
+ "step": 5960
+ },
+ {
+ "epoch": 1.6032778094312432,
+ "grad_norm": 0.1916576772928238,
+ "learning_rate": 2.977882037533512e-06,
+ "loss": 0.4859332084655762,
+ "step": 5980
+ },
+ {
+ "epoch": 1.6086402788484095,
+ "grad_norm": 0.23635101318359375,
+ "learning_rate": 2.9376675603217157e-06,
+ "loss": 0.5418910980224609,
+ "step": 6000
+ },
+ {
+ "epoch": 1.6140027482655763,
+ "grad_norm": 0.2404562532901764,
+ "learning_rate": 2.89745308310992e-06,
+ "loss": 0.5449445247650146,
+ "step": 6020
+ },
+ {
+ "epoch": 1.6193652176827429,
+ "grad_norm": 0.20147347450256348,
+ "learning_rate": 2.8572386058981235e-06,
+ "loss": 0.4737790584564209,
+ "step": 6040
+ },
+ {
+ "epoch": 1.6247276870999094,
+ "grad_norm": 0.2455863654613495,
+ "learning_rate": 2.8170241286863272e-06,
+ "loss": 0.4722298145294189,
+ "step": 6060
+ },
+ {
+ "epoch": 1.6300901565170762,
+ "grad_norm": 0.22172148525714874,
+ "learning_rate": 2.776809651474531e-06,
+ "loss": 0.5120372295379638,
+ "step": 6080
+ },
+ {
+ "epoch": 1.6354526259342426,
+ "grad_norm": 0.3848462700843811,
+ "learning_rate": 2.7365951742627346e-06,
+ "loss": 0.5152206897735596,
+ "step": 6100
+ },
+ {
+ "epoch": 1.6408150953514093,
+ "grad_norm": 0.19071047008037567,
+ "learning_rate": 2.6963806970509383e-06,
+ "loss": 0.4757692813873291,
+ "step": 6120
+ },
+ {
+ "epoch": 1.646177564768576,
+ "grad_norm": 0.20568661391735077,
+ "learning_rate": 2.656166219839142e-06,
+ "loss": 0.475917387008667,
+ "step": 6140
+ },
+ {
+ "epoch": 1.6515400341857425,
+ "grad_norm": 0.11777322739362717,
+ "learning_rate": 2.6159517426273457e-06,
+ "loss": 0.5161296367645264,
+ "step": 6160
+ },
+ {
+ "epoch": 1.6569025036029092,
+ "grad_norm": 0.1700555831193924,
+ "learning_rate": 2.5757372654155494e-06,
+ "loss": 0.4715432167053223,
+ "step": 6180
+ },
+ {
+ "epoch": 1.6622649730200756,
+ "grad_norm": 0.18927083909511566,
+ "learning_rate": 2.5355227882037535e-06,
+ "loss": 0.49937710762023924,
+ "step": 6200
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 7.631987064833311e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-6200/training_args.bin b/checkpoint-6200/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-6200/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-6400/README.md b/checkpoint-6400/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-6400/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-6400/adapter_config.json b/checkpoint-6400/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-6400/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-6400/adapter_model.safetensors b/checkpoint-6400/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..a63085d98b2fa346a94c8ce501dc1ca67eac67af
--- /dev/null
+++ b/checkpoint-6400/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:449b3b4c9fca8ac1f4f4392005b516f91382fe9d2311f3675a9d1ca8f9081f68
+size 50360752
diff --git a/checkpoint-6400/chat_template.jinja b/checkpoint-6400/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-6400/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-6400/optimizer.pt b/checkpoint-6400/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..601e5f97b282ab0632a0f8498ceff8bce151d044
--- /dev/null
+++ b/checkpoint-6400/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:abce19518619687e5a1c0090a291dd0ad478c935d4ad81a66becd5c451f30aa5
+size 100828235
diff --git a/checkpoint-6400/rng_state.pth b/checkpoint-6400/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..3ee1dd1b014381d33facdb6549fb6b59b2faa38c
--- /dev/null
+++ b/checkpoint-6400/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d90b097f445f1e554b6312c9ebf2c9fa0b31dc51bedadaa8f8fbc4a3864d8a2f
+size 14645
diff --git a/checkpoint-6400/scheduler.pt b/checkpoint-6400/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..2aa35e9b8f442b2db543507ca4bd17fbb0ae4efc
--- /dev/null
+++ b/checkpoint-6400/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:9e2add601d450aaa7de8db6075b3038505d99dd0b0cab0c4e047224526f466c7
+size 1465
diff --git a/checkpoint-6400/tokenizer.json b/checkpoint-6400/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-6400/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-6400/tokenizer_config.json b/checkpoint-6400/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-6400/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-6400/trainer_state.json b/checkpoint-6400/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..f57595567d78d9babdb1a0b3d9d7ea43cb260923
--- /dev/null
+++ b/checkpoint-6400/trainer_state.json
@@ -0,0 +1,2274 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.7158896671917419,
+ "eval_steps": 500,
+ "global_step": 6400,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ },
+ {
+ "epoch": 1.185005194892248,
+ "grad_norm": 0.16667844355106354,
+ "learning_rate": 6.114611260053619e-06,
+ "loss": 0.5159809589385986,
+ "step": 4420
+ },
+ {
+ "epoch": 1.1903676643094145,
+ "grad_norm": 0.19206570088863373,
+ "learning_rate": 6.074396782841823e-06,
+ "loss": 0.47541089057922364,
+ "step": 4440
+ },
+ {
+ "epoch": 1.195730133726581,
+ "grad_norm": 0.17394617199897766,
+ "learning_rate": 6.034182305630027e-06,
+ "loss": 0.5470661640167236,
+ "step": 4460
+ },
+ {
+ "epoch": 1.2010926031437477,
+ "grad_norm": 0.210404634475708,
+ "learning_rate": 5.993967828418231e-06,
+ "loss": 0.5377882957458496,
+ "step": 4480
+ },
+ {
+ "epoch": 1.2064550725609142,
+ "grad_norm": 0.18084648251533508,
+ "learning_rate": 5.953753351206435e-06,
+ "loss": 0.5037185192108155,
+ "step": 4500
+ },
+ {
+ "epoch": 1.211817541978081,
+ "grad_norm": 0.23707027733325958,
+ "learning_rate": 5.913538873994639e-06,
+ "loss": 0.4822190284729004,
+ "step": 4520
+ },
+ {
+ "epoch": 1.2171800113952476,
+ "grad_norm": 0.16474473476409912,
+ "learning_rate": 5.873324396782842e-06,
+ "loss": 0.46645288467407225,
+ "step": 4540
+ },
+ {
+ "epoch": 1.2225424808124141,
+ "grad_norm": 0.2142348438501358,
+ "learning_rate": 5.833109919571046e-06,
+ "loss": 0.5255855560302735,
+ "step": 4560
+ },
+ {
+ "epoch": 1.2279049502295807,
+ "grad_norm": 0.2531765103340149,
+ "learning_rate": 5.79289544235925e-06,
+ "loss": 0.507044792175293,
+ "step": 4580
+ },
+ {
+ "epoch": 1.2332674196467472,
+ "grad_norm": 0.2553550899028778,
+ "learning_rate": 5.7526809651474535e-06,
+ "loss": 0.4767824649810791,
+ "step": 4600
+ },
+ {
+ "epoch": 1.238629889063914,
+ "grad_norm": 0.14484412968158722,
+ "learning_rate": 5.712466487935657e-06,
+ "loss": 0.4675601005554199,
+ "step": 4620
+ },
+ {
+ "epoch": 1.2439923584810806,
+ "grad_norm": 0.14328251779079437,
+ "learning_rate": 5.672252010723861e-06,
+ "loss": 0.4956005573272705,
+ "step": 4640
+ },
+ {
+ "epoch": 1.2493548278982471,
+ "grad_norm": 0.1739245355129242,
+ "learning_rate": 5.632037533512065e-06,
+ "loss": 0.48583345413208007,
+ "step": 4660
+ },
+ {
+ "epoch": 1.2547172973154137,
+ "grad_norm": 0.21294184029102325,
+ "learning_rate": 5.591823056300268e-06,
+ "loss": 0.520921277999878,
+ "step": 4680
+ },
+ {
+ "epoch": 1.2600797667325803,
+ "grad_norm": 0.25132355093955994,
+ "learning_rate": 5.551608579088472e-06,
+ "loss": 0.5295385837554931,
+ "step": 4700
+ },
+ {
+ "epoch": 1.265442236149747,
+ "grad_norm": 0.18603841960430145,
+ "learning_rate": 5.511394101876676e-06,
+ "loss": 0.47570199966430665,
+ "step": 4720
+ },
+ {
+ "epoch": 1.2708047055669136,
+ "grad_norm": 0.19883134961128235,
+ "learning_rate": 5.471179624664879e-06,
+ "loss": 0.5016080379486084,
+ "step": 4740
+ },
+ {
+ "epoch": 1.2761671749840802,
+ "grad_norm": 0.19640181958675385,
+ "learning_rate": 5.430965147453083e-06,
+ "loss": 0.4999081134796143,
+ "step": 4760
+ },
+ {
+ "epoch": 1.2815296444012467,
+ "grad_norm": 0.2584764361381531,
+ "learning_rate": 5.390750670241287e-06,
+ "loss": 0.4780082702636719,
+ "step": 4780
+ },
+ {
+ "epoch": 1.2868921138184133,
+ "grad_norm": 0.2925741374492645,
+ "learning_rate": 5.3505361930294905e-06,
+ "loss": 0.5131395816802978,
+ "step": 4800
+ },
+ {
+ "epoch": 1.29225458323558,
+ "grad_norm": 0.18971531093120575,
+ "learning_rate": 5.310321715817694e-06,
+ "loss": 0.455674409866333,
+ "step": 4820
+ },
+ {
+ "epoch": 1.2976170526527466,
+ "grad_norm": 0.16778405010700226,
+ "learning_rate": 5.270107238605898e-06,
+ "loss": 0.5070962905883789,
+ "step": 4840
+ },
+ {
+ "epoch": 1.3029795220699132,
+ "grad_norm": 0.30026957392692566,
+ "learning_rate": 5.2298927613941016e-06,
+ "loss": 0.5120027542114258,
+ "step": 4860
+ },
+ {
+ "epoch": 1.3083419914870797,
+ "grad_norm": 0.17846634984016418,
+ "learning_rate": 5.189678284182305e-06,
+ "loss": 0.5114477157592774,
+ "step": 4880
+ },
+ {
+ "epoch": 1.3137044609042463,
+ "grad_norm": 0.1962418258190155,
+ "learning_rate": 5.149463806970509e-06,
+ "loss": 0.5043613910675049,
+ "step": 4900
+ },
+ {
+ "epoch": 1.319066930321413,
+ "grad_norm": 0.18446756899356842,
+ "learning_rate": 5.1092493297587135e-06,
+ "loss": 0.5396455287933349,
+ "step": 4920
+ },
+ {
+ "epoch": 1.3244293997385796,
+ "grad_norm": 0.20886844396591187,
+ "learning_rate": 5.069034852546917e-06,
+ "loss": 0.4879767417907715,
+ "step": 4940
+ },
+ {
+ "epoch": 1.3297918691557462,
+ "grad_norm": 0.16687901318073273,
+ "learning_rate": 5.028820375335121e-06,
+ "loss": 0.5014327049255372,
+ "step": 4960
+ },
+ {
+ "epoch": 1.3351543385729128,
+ "grad_norm": 0.19595153629779816,
+ "learning_rate": 4.988605898123325e-06,
+ "loss": 0.5375277996063232,
+ "step": 4980
+ },
+ {
+ "epoch": 1.3405168079900793,
+ "grad_norm": 0.2372344732284546,
+ "learning_rate": 4.948391420911528e-06,
+ "loss": 0.5020076274871826,
+ "step": 5000
+ },
+ {
+ "epoch": 1.345879277407246,
+ "grad_norm": 0.21030014753341675,
+ "learning_rate": 4.908176943699732e-06,
+ "loss": 0.5111066818237304,
+ "step": 5020
+ },
+ {
+ "epoch": 1.3512417468244127,
+ "grad_norm": 0.1866692751646042,
+ "learning_rate": 4.867962466487936e-06,
+ "loss": 0.4515383720397949,
+ "step": 5040
+ },
+ {
+ "epoch": 1.3566042162415792,
+ "grad_norm": 0.22531798481941223,
+ "learning_rate": 4.827747989276139e-06,
+ "loss": 0.4757690906524658,
+ "step": 5060
+ },
+ {
+ "epoch": 1.3619666856587458,
+ "grad_norm": 0.15868768095970154,
+ "learning_rate": 4.787533512064343e-06,
+ "loss": 0.45842318534851073,
+ "step": 5080
+ },
+ {
+ "epoch": 1.3673291550759124,
+ "grad_norm": 0.24528546631336212,
+ "learning_rate": 4.747319034852547e-06,
+ "loss": 0.47269258499145506,
+ "step": 5100
+ },
+ {
+ "epoch": 1.3726916244930791,
+ "grad_norm": 0.17387732863426208,
+ "learning_rate": 4.707104557640751e-06,
+ "loss": 0.5103805065155029,
+ "step": 5120
+ },
+ {
+ "epoch": 1.3780540939102457,
+ "grad_norm": 0.20686905086040497,
+ "learning_rate": 4.666890080428955e-06,
+ "loss": 0.5135180950164795,
+ "step": 5140
+ },
+ {
+ "epoch": 1.3834165633274123,
+ "grad_norm": 0.19599783420562744,
+ "learning_rate": 4.626675603217159e-06,
+ "loss": 0.5045839786529541,
+ "step": 5160
+ },
+ {
+ "epoch": 1.3887790327445788,
+ "grad_norm": 0.2585010528564453,
+ "learning_rate": 4.586461126005362e-06,
+ "loss": 0.45903496742248534,
+ "step": 5180
+ },
+ {
+ "epoch": 1.3941415021617454,
+ "grad_norm": 0.1688319593667984,
+ "learning_rate": 4.546246648793566e-06,
+ "loss": 0.5017509937286377,
+ "step": 5200
+ },
+ {
+ "epoch": 1.3995039715789122,
+ "grad_norm": 0.21520815789699554,
+ "learning_rate": 4.50603217158177e-06,
+ "loss": 0.48459539413452146,
+ "step": 5220
+ },
+ {
+ "epoch": 1.4048664409960787,
+ "grad_norm": 0.20514647662639618,
+ "learning_rate": 4.4658176943699735e-06,
+ "loss": 0.5073423862457276,
+ "step": 5240
+ },
+ {
+ "epoch": 1.4102289104132453,
+ "grad_norm": 0.21835413575172424,
+ "learning_rate": 4.425603217158177e-06,
+ "loss": 0.5290310382843018,
+ "step": 5260
+ },
+ {
+ "epoch": 1.4155913798304118,
+ "grad_norm": 0.28042587637901306,
+ "learning_rate": 4.385388739946381e-06,
+ "loss": 0.4823312759399414,
+ "step": 5280
+ },
+ {
+ "epoch": 1.4209538492475784,
+ "grad_norm": 0.18959026038646698,
+ "learning_rate": 4.345174262734585e-06,
+ "loss": 0.4921241760253906,
+ "step": 5300
+ },
+ {
+ "epoch": 1.4263163186647452,
+ "grad_norm": 0.18584316968917847,
+ "learning_rate": 4.304959785522788e-06,
+ "loss": 0.4892130374908447,
+ "step": 5320
+ },
+ {
+ "epoch": 1.4316787880819117,
+ "grad_norm": 0.17588038742542267,
+ "learning_rate": 4.264745308310992e-06,
+ "loss": 0.4822041988372803,
+ "step": 5340
+ },
+ {
+ "epoch": 1.4370412574990783,
+ "grad_norm": 0.18146033585071564,
+ "learning_rate": 4.224530831099196e-06,
+ "loss": 0.5084807395935058,
+ "step": 5360
+ },
+ {
+ "epoch": 1.4424037269162449,
+ "grad_norm": 0.2251797467470169,
+ "learning_rate": 4.184316353887399e-06,
+ "loss": 0.5146170139312745,
+ "step": 5380
+ },
+ {
+ "epoch": 1.4477661963334114,
+ "grad_norm": 0.18744796514511108,
+ "learning_rate": 4.144101876675603e-06,
+ "loss": 0.5189927577972412,
+ "step": 5400
+ },
+ {
+ "epoch": 1.4531286657505782,
+ "grad_norm": 0.25737133622169495,
+ "learning_rate": 4.103887399463807e-06,
+ "loss": 0.4891658782958984,
+ "step": 5420
+ },
+ {
+ "epoch": 1.4584911351677448,
+ "grad_norm": 0.20580479502677917,
+ "learning_rate": 4.0636729222520105e-06,
+ "loss": 0.4953591823577881,
+ "step": 5440
+ },
+ {
+ "epoch": 1.4638536045849113,
+ "grad_norm": 0.2351546287536621,
+ "learning_rate": 4.023458445040214e-06,
+ "loss": 0.5025320053100586,
+ "step": 5460
+ },
+ {
+ "epoch": 1.4692160740020779,
+ "grad_norm": 0.1819481998682022,
+ "learning_rate": 3.983243967828418e-06,
+ "loss": 0.47151756286621094,
+ "step": 5480
+ },
+ {
+ "epoch": 1.4745785434192444,
+ "grad_norm": 0.20772472023963928,
+ "learning_rate": 3.943029490616622e-06,
+ "loss": 0.4678915023803711,
+ "step": 5500
+ },
+ {
+ "epoch": 1.4799410128364112,
+ "grad_norm": 0.2203037440776825,
+ "learning_rate": 3.902815013404825e-06,
+ "loss": 0.46007452011108396,
+ "step": 5520
+ },
+ {
+ "epoch": 1.4853034822535778,
+ "grad_norm": 0.15371400117874146,
+ "learning_rate": 3.86260053619303e-06,
+ "loss": 0.44407024383544924,
+ "step": 5540
+ },
+ {
+ "epoch": 1.4906659516707443,
+ "grad_norm": 0.2276080846786499,
+ "learning_rate": 3.8223860589812335e-06,
+ "loss": 0.4730556488037109,
+ "step": 5560
+ },
+ {
+ "epoch": 1.4960284210879111,
+ "grad_norm": 0.24482466280460358,
+ "learning_rate": 3.7821715817694376e-06,
+ "loss": 0.5073911666870117,
+ "step": 5580
+ },
+ {
+ "epoch": 1.5013908905050775,
+ "grad_norm": 0.20438458025455475,
+ "learning_rate": 3.741957104557641e-06,
+ "loss": 0.46701641082763673,
+ "step": 5600
+ },
+ {
+ "epoch": 1.5067533599222442,
+ "grad_norm": 0.19854313135147095,
+ "learning_rate": 3.7017426273458446e-06,
+ "loss": 0.46309399604797363,
+ "step": 5620
+ },
+ {
+ "epoch": 1.5121158293394108,
+ "grad_norm": 0.18356069922447205,
+ "learning_rate": 3.6615281501340483e-06,
+ "loss": 0.503613805770874,
+ "step": 5640
+ },
+ {
+ "epoch": 1.5174782987565774,
+ "grad_norm": 0.2009744495153427,
+ "learning_rate": 3.621313672922252e-06,
+ "loss": 0.4765054225921631,
+ "step": 5660
+ },
+ {
+ "epoch": 1.5228407681737441,
+ "grad_norm": 0.3058745563030243,
+ "learning_rate": 3.5810991957104557e-06,
+ "loss": 0.5179148197174073,
+ "step": 5680
+ },
+ {
+ "epoch": 1.5282032375909105,
+ "grad_norm": 0.17671597003936768,
+ "learning_rate": 3.54088471849866e-06,
+ "loss": 0.45907344818115237,
+ "step": 5700
+ },
+ {
+ "epoch": 1.5335657070080773,
+ "grad_norm": 0.22209160029888153,
+ "learning_rate": 3.5006702412868635e-06,
+ "loss": 0.49304862022399903,
+ "step": 5720
+ },
+ {
+ "epoch": 1.5389281764252438,
+ "grad_norm": 0.21018914878368378,
+ "learning_rate": 3.4604557640750672e-06,
+ "loss": 0.5536758422851562,
+ "step": 5740
+ },
+ {
+ "epoch": 1.5442906458424104,
+ "grad_norm": 0.14339996874332428,
+ "learning_rate": 3.420241286863271e-06,
+ "loss": 0.48726091384887693,
+ "step": 5760
+ },
+ {
+ "epoch": 1.5496531152595772,
+ "grad_norm": 0.11419746279716492,
+ "learning_rate": 3.3800268096514746e-06,
+ "loss": 0.4514151573181152,
+ "step": 5780
+ },
+ {
+ "epoch": 1.5550155846767435,
+ "grad_norm": 0.18168962001800537,
+ "learning_rate": 3.3398123324396783e-06,
+ "loss": 0.5279990196228027,
+ "step": 5800
+ },
+ {
+ "epoch": 1.5603780540939103,
+ "grad_norm": 0.24244488775730133,
+ "learning_rate": 3.299597855227882e-06,
+ "loss": 0.49297361373901366,
+ "step": 5820
+ },
+ {
+ "epoch": 1.5657405235110768,
+ "grad_norm": 0.2017296999692917,
+ "learning_rate": 3.2593833780160857e-06,
+ "loss": 0.49305019378662107,
+ "step": 5840
+ },
+ {
+ "epoch": 1.5711029929282434,
+ "grad_norm": 0.22592377662658691,
+ "learning_rate": 3.2191689008042894e-06,
+ "loss": 0.4862989902496338,
+ "step": 5860
+ },
+ {
+ "epoch": 1.5764654623454102,
+ "grad_norm": 0.24772357940673828,
+ "learning_rate": 3.1789544235924935e-06,
+ "loss": 0.45182647705078127,
+ "step": 5880
+ },
+ {
+ "epoch": 1.5818279317625765,
+ "grad_norm": 0.20607218146324158,
+ "learning_rate": 3.1387399463806972e-06,
+ "loss": 0.48905248641967775,
+ "step": 5900
+ },
+ {
+ "epoch": 1.5871904011797433,
+ "grad_norm": 0.1931353509426117,
+ "learning_rate": 3.098525469168901e-06,
+ "loss": 0.5307461261749268,
+ "step": 5920
+ },
+ {
+ "epoch": 1.5925528705969099,
+ "grad_norm": 0.16020581126213074,
+ "learning_rate": 3.0583109919571046e-06,
+ "loss": 0.4672811985015869,
+ "step": 5940
+ },
+ {
+ "epoch": 1.5979153400140764,
+ "grad_norm": 0.23668015003204346,
+ "learning_rate": 3.0180965147453083e-06,
+ "loss": 0.5272688865661621,
+ "step": 5960
+ },
+ {
+ "epoch": 1.6032778094312432,
+ "grad_norm": 0.1916576772928238,
+ "learning_rate": 2.977882037533512e-06,
+ "loss": 0.4859332084655762,
+ "step": 5980
+ },
+ {
+ "epoch": 1.6086402788484095,
+ "grad_norm": 0.23635101318359375,
+ "learning_rate": 2.9376675603217157e-06,
+ "loss": 0.5418910980224609,
+ "step": 6000
+ },
+ {
+ "epoch": 1.6140027482655763,
+ "grad_norm": 0.2404562532901764,
+ "learning_rate": 2.89745308310992e-06,
+ "loss": 0.5449445247650146,
+ "step": 6020
+ },
+ {
+ "epoch": 1.6193652176827429,
+ "grad_norm": 0.20147347450256348,
+ "learning_rate": 2.8572386058981235e-06,
+ "loss": 0.4737790584564209,
+ "step": 6040
+ },
+ {
+ "epoch": 1.6247276870999094,
+ "grad_norm": 0.2455863654613495,
+ "learning_rate": 2.8170241286863272e-06,
+ "loss": 0.4722298145294189,
+ "step": 6060
+ },
+ {
+ "epoch": 1.6300901565170762,
+ "grad_norm": 0.22172148525714874,
+ "learning_rate": 2.776809651474531e-06,
+ "loss": 0.5120372295379638,
+ "step": 6080
+ },
+ {
+ "epoch": 1.6354526259342426,
+ "grad_norm": 0.3848462700843811,
+ "learning_rate": 2.7365951742627346e-06,
+ "loss": 0.5152206897735596,
+ "step": 6100
+ },
+ {
+ "epoch": 1.6408150953514093,
+ "grad_norm": 0.19071047008037567,
+ "learning_rate": 2.6963806970509383e-06,
+ "loss": 0.4757692813873291,
+ "step": 6120
+ },
+ {
+ "epoch": 1.646177564768576,
+ "grad_norm": 0.20568661391735077,
+ "learning_rate": 2.656166219839142e-06,
+ "loss": 0.475917387008667,
+ "step": 6140
+ },
+ {
+ "epoch": 1.6515400341857425,
+ "grad_norm": 0.11777322739362717,
+ "learning_rate": 2.6159517426273457e-06,
+ "loss": 0.5161296367645264,
+ "step": 6160
+ },
+ {
+ "epoch": 1.6569025036029092,
+ "grad_norm": 0.1700555831193924,
+ "learning_rate": 2.5757372654155494e-06,
+ "loss": 0.4715432167053223,
+ "step": 6180
+ },
+ {
+ "epoch": 1.6622649730200756,
+ "grad_norm": 0.18927083909511566,
+ "learning_rate": 2.5355227882037535e-06,
+ "loss": 0.49937710762023924,
+ "step": 6200
+ },
+ {
+ "epoch": 1.6676274424372424,
+ "grad_norm": 0.22097784280776978,
+ "learning_rate": 2.4953083109919572e-06,
+ "loss": 0.43366107940673826,
+ "step": 6220
+ },
+ {
+ "epoch": 1.672989911854409,
+ "grad_norm": 0.2299281805753708,
+ "learning_rate": 2.455093833780161e-06,
+ "loss": 0.5145821094512939,
+ "step": 6240
+ },
+ {
+ "epoch": 1.6783523812715755,
+ "grad_norm": 0.2384844720363617,
+ "learning_rate": 2.4148793565683646e-06,
+ "loss": 0.459308385848999,
+ "step": 6260
+ },
+ {
+ "epoch": 1.6837148506887423,
+ "grad_norm": 0.24471035599708557,
+ "learning_rate": 2.3746648793565683e-06,
+ "loss": 0.4676504611968994,
+ "step": 6280
+ },
+ {
+ "epoch": 1.6890773201059086,
+ "grad_norm": 0.24419866502285004,
+ "learning_rate": 2.334450402144772e-06,
+ "loss": 0.4745138168334961,
+ "step": 6300
+ },
+ {
+ "epoch": 1.6944397895230754,
+ "grad_norm": 0.15896575152873993,
+ "learning_rate": 2.294235924932976e-06,
+ "loss": 0.5073649883270264,
+ "step": 6320
+ },
+ {
+ "epoch": 1.699802258940242,
+ "grad_norm": 0.26504868268966675,
+ "learning_rate": 2.25402144772118e-06,
+ "loss": 0.4534353733062744,
+ "step": 6340
+ },
+ {
+ "epoch": 1.7051647283574085,
+ "grad_norm": 0.2461850792169571,
+ "learning_rate": 2.2138069705093836e-06,
+ "loss": 0.4862947940826416,
+ "step": 6360
+ },
+ {
+ "epoch": 1.7105271977745753,
+ "grad_norm": 0.17332817614078522,
+ "learning_rate": 2.1735924932975873e-06,
+ "loss": 0.5049370765686035,
+ "step": 6380
+ },
+ {
+ "epoch": 1.7158896671917419,
+ "grad_norm": 0.19762548804283142,
+ "learning_rate": 2.133378016085791e-06,
+ "loss": 0.5272616386413574,
+ "step": 6400
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 7.880667884237722e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-6400/training_args.bin b/checkpoint-6400/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-6400/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-6600/README.md b/checkpoint-6600/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-6600/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-6600/adapter_config.json b/checkpoint-6600/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-6600/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-6600/adapter_model.safetensors b/checkpoint-6600/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..5f09d27c16c9d9057d6be385d0d158b846b7e3e0
--- /dev/null
+++ b/checkpoint-6600/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:07858729b9755288cd707add485cd622a969df7813c4e7957784677567fde546
+size 50360752
diff --git a/checkpoint-6600/chat_template.jinja b/checkpoint-6600/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-6600/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-6600/optimizer.pt b/checkpoint-6600/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..52e7c940290592ea4b76ac347c41d62375454d64
--- /dev/null
+++ b/checkpoint-6600/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:9c8e43a0540ac047a9fcc2c58c25d2bb53c1b8a83fc85ceb99d55eebd6e04093
+size 100828235
diff --git a/checkpoint-6600/rng_state.pth b/checkpoint-6600/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..563750c0255da92880e11402fc1b2dadfc0b5d65
--- /dev/null
+++ b/checkpoint-6600/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:16843b9ee8bbfd86b883e92d6739e6aa0aa96746771df4a0ef0f3191b38210b7
+size 14645
diff --git a/checkpoint-6600/scheduler.pt b/checkpoint-6600/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..f6a5e4822738abe47a37d12a83cd5178a2bc0574
--- /dev/null
+++ b/checkpoint-6600/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:25bfade1230c63a6e64bbdacfdf5f14af936aeb3654bdd194db91fbf8f3883cd
+size 1465
diff --git a/checkpoint-6600/tokenizer.json b/checkpoint-6600/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-6600/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-6600/tokenizer_config.json b/checkpoint-6600/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-6600/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-6600/trainer_state.json b/checkpoint-6600/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..a6f2244bcc9bba8f26556b3462a1f3d9b10fb0a7
--- /dev/null
+++ b/checkpoint-6600/trainer_state.json
@@ -0,0 +1,2344 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.769514361363408,
+ "eval_steps": 500,
+ "global_step": 6600,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ },
+ {
+ "epoch": 1.185005194892248,
+ "grad_norm": 0.16667844355106354,
+ "learning_rate": 6.114611260053619e-06,
+ "loss": 0.5159809589385986,
+ "step": 4420
+ },
+ {
+ "epoch": 1.1903676643094145,
+ "grad_norm": 0.19206570088863373,
+ "learning_rate": 6.074396782841823e-06,
+ "loss": 0.47541089057922364,
+ "step": 4440
+ },
+ {
+ "epoch": 1.195730133726581,
+ "grad_norm": 0.17394617199897766,
+ "learning_rate": 6.034182305630027e-06,
+ "loss": 0.5470661640167236,
+ "step": 4460
+ },
+ {
+ "epoch": 1.2010926031437477,
+ "grad_norm": 0.210404634475708,
+ "learning_rate": 5.993967828418231e-06,
+ "loss": 0.5377882957458496,
+ "step": 4480
+ },
+ {
+ "epoch": 1.2064550725609142,
+ "grad_norm": 0.18084648251533508,
+ "learning_rate": 5.953753351206435e-06,
+ "loss": 0.5037185192108155,
+ "step": 4500
+ },
+ {
+ "epoch": 1.211817541978081,
+ "grad_norm": 0.23707027733325958,
+ "learning_rate": 5.913538873994639e-06,
+ "loss": 0.4822190284729004,
+ "step": 4520
+ },
+ {
+ "epoch": 1.2171800113952476,
+ "grad_norm": 0.16474473476409912,
+ "learning_rate": 5.873324396782842e-06,
+ "loss": 0.46645288467407225,
+ "step": 4540
+ },
+ {
+ "epoch": 1.2225424808124141,
+ "grad_norm": 0.2142348438501358,
+ "learning_rate": 5.833109919571046e-06,
+ "loss": 0.5255855560302735,
+ "step": 4560
+ },
+ {
+ "epoch": 1.2279049502295807,
+ "grad_norm": 0.2531765103340149,
+ "learning_rate": 5.79289544235925e-06,
+ "loss": 0.507044792175293,
+ "step": 4580
+ },
+ {
+ "epoch": 1.2332674196467472,
+ "grad_norm": 0.2553550899028778,
+ "learning_rate": 5.7526809651474535e-06,
+ "loss": 0.4767824649810791,
+ "step": 4600
+ },
+ {
+ "epoch": 1.238629889063914,
+ "grad_norm": 0.14484412968158722,
+ "learning_rate": 5.712466487935657e-06,
+ "loss": 0.4675601005554199,
+ "step": 4620
+ },
+ {
+ "epoch": 1.2439923584810806,
+ "grad_norm": 0.14328251779079437,
+ "learning_rate": 5.672252010723861e-06,
+ "loss": 0.4956005573272705,
+ "step": 4640
+ },
+ {
+ "epoch": 1.2493548278982471,
+ "grad_norm": 0.1739245355129242,
+ "learning_rate": 5.632037533512065e-06,
+ "loss": 0.48583345413208007,
+ "step": 4660
+ },
+ {
+ "epoch": 1.2547172973154137,
+ "grad_norm": 0.21294184029102325,
+ "learning_rate": 5.591823056300268e-06,
+ "loss": 0.520921277999878,
+ "step": 4680
+ },
+ {
+ "epoch": 1.2600797667325803,
+ "grad_norm": 0.25132355093955994,
+ "learning_rate": 5.551608579088472e-06,
+ "loss": 0.5295385837554931,
+ "step": 4700
+ },
+ {
+ "epoch": 1.265442236149747,
+ "grad_norm": 0.18603841960430145,
+ "learning_rate": 5.511394101876676e-06,
+ "loss": 0.47570199966430665,
+ "step": 4720
+ },
+ {
+ "epoch": 1.2708047055669136,
+ "grad_norm": 0.19883134961128235,
+ "learning_rate": 5.471179624664879e-06,
+ "loss": 0.5016080379486084,
+ "step": 4740
+ },
+ {
+ "epoch": 1.2761671749840802,
+ "grad_norm": 0.19640181958675385,
+ "learning_rate": 5.430965147453083e-06,
+ "loss": 0.4999081134796143,
+ "step": 4760
+ },
+ {
+ "epoch": 1.2815296444012467,
+ "grad_norm": 0.2584764361381531,
+ "learning_rate": 5.390750670241287e-06,
+ "loss": 0.4780082702636719,
+ "step": 4780
+ },
+ {
+ "epoch": 1.2868921138184133,
+ "grad_norm": 0.2925741374492645,
+ "learning_rate": 5.3505361930294905e-06,
+ "loss": 0.5131395816802978,
+ "step": 4800
+ },
+ {
+ "epoch": 1.29225458323558,
+ "grad_norm": 0.18971531093120575,
+ "learning_rate": 5.310321715817694e-06,
+ "loss": 0.455674409866333,
+ "step": 4820
+ },
+ {
+ "epoch": 1.2976170526527466,
+ "grad_norm": 0.16778405010700226,
+ "learning_rate": 5.270107238605898e-06,
+ "loss": 0.5070962905883789,
+ "step": 4840
+ },
+ {
+ "epoch": 1.3029795220699132,
+ "grad_norm": 0.30026957392692566,
+ "learning_rate": 5.2298927613941016e-06,
+ "loss": 0.5120027542114258,
+ "step": 4860
+ },
+ {
+ "epoch": 1.3083419914870797,
+ "grad_norm": 0.17846634984016418,
+ "learning_rate": 5.189678284182305e-06,
+ "loss": 0.5114477157592774,
+ "step": 4880
+ },
+ {
+ "epoch": 1.3137044609042463,
+ "grad_norm": 0.1962418258190155,
+ "learning_rate": 5.149463806970509e-06,
+ "loss": 0.5043613910675049,
+ "step": 4900
+ },
+ {
+ "epoch": 1.319066930321413,
+ "grad_norm": 0.18446756899356842,
+ "learning_rate": 5.1092493297587135e-06,
+ "loss": 0.5396455287933349,
+ "step": 4920
+ },
+ {
+ "epoch": 1.3244293997385796,
+ "grad_norm": 0.20886844396591187,
+ "learning_rate": 5.069034852546917e-06,
+ "loss": 0.4879767417907715,
+ "step": 4940
+ },
+ {
+ "epoch": 1.3297918691557462,
+ "grad_norm": 0.16687901318073273,
+ "learning_rate": 5.028820375335121e-06,
+ "loss": 0.5014327049255372,
+ "step": 4960
+ },
+ {
+ "epoch": 1.3351543385729128,
+ "grad_norm": 0.19595153629779816,
+ "learning_rate": 4.988605898123325e-06,
+ "loss": 0.5375277996063232,
+ "step": 4980
+ },
+ {
+ "epoch": 1.3405168079900793,
+ "grad_norm": 0.2372344732284546,
+ "learning_rate": 4.948391420911528e-06,
+ "loss": 0.5020076274871826,
+ "step": 5000
+ },
+ {
+ "epoch": 1.345879277407246,
+ "grad_norm": 0.21030014753341675,
+ "learning_rate": 4.908176943699732e-06,
+ "loss": 0.5111066818237304,
+ "step": 5020
+ },
+ {
+ "epoch": 1.3512417468244127,
+ "grad_norm": 0.1866692751646042,
+ "learning_rate": 4.867962466487936e-06,
+ "loss": 0.4515383720397949,
+ "step": 5040
+ },
+ {
+ "epoch": 1.3566042162415792,
+ "grad_norm": 0.22531798481941223,
+ "learning_rate": 4.827747989276139e-06,
+ "loss": 0.4757690906524658,
+ "step": 5060
+ },
+ {
+ "epoch": 1.3619666856587458,
+ "grad_norm": 0.15868768095970154,
+ "learning_rate": 4.787533512064343e-06,
+ "loss": 0.45842318534851073,
+ "step": 5080
+ },
+ {
+ "epoch": 1.3673291550759124,
+ "grad_norm": 0.24528546631336212,
+ "learning_rate": 4.747319034852547e-06,
+ "loss": 0.47269258499145506,
+ "step": 5100
+ },
+ {
+ "epoch": 1.3726916244930791,
+ "grad_norm": 0.17387732863426208,
+ "learning_rate": 4.707104557640751e-06,
+ "loss": 0.5103805065155029,
+ "step": 5120
+ },
+ {
+ "epoch": 1.3780540939102457,
+ "grad_norm": 0.20686905086040497,
+ "learning_rate": 4.666890080428955e-06,
+ "loss": 0.5135180950164795,
+ "step": 5140
+ },
+ {
+ "epoch": 1.3834165633274123,
+ "grad_norm": 0.19599783420562744,
+ "learning_rate": 4.626675603217159e-06,
+ "loss": 0.5045839786529541,
+ "step": 5160
+ },
+ {
+ "epoch": 1.3887790327445788,
+ "grad_norm": 0.2585010528564453,
+ "learning_rate": 4.586461126005362e-06,
+ "loss": 0.45903496742248534,
+ "step": 5180
+ },
+ {
+ "epoch": 1.3941415021617454,
+ "grad_norm": 0.1688319593667984,
+ "learning_rate": 4.546246648793566e-06,
+ "loss": 0.5017509937286377,
+ "step": 5200
+ },
+ {
+ "epoch": 1.3995039715789122,
+ "grad_norm": 0.21520815789699554,
+ "learning_rate": 4.50603217158177e-06,
+ "loss": 0.48459539413452146,
+ "step": 5220
+ },
+ {
+ "epoch": 1.4048664409960787,
+ "grad_norm": 0.20514647662639618,
+ "learning_rate": 4.4658176943699735e-06,
+ "loss": 0.5073423862457276,
+ "step": 5240
+ },
+ {
+ "epoch": 1.4102289104132453,
+ "grad_norm": 0.21835413575172424,
+ "learning_rate": 4.425603217158177e-06,
+ "loss": 0.5290310382843018,
+ "step": 5260
+ },
+ {
+ "epoch": 1.4155913798304118,
+ "grad_norm": 0.28042587637901306,
+ "learning_rate": 4.385388739946381e-06,
+ "loss": 0.4823312759399414,
+ "step": 5280
+ },
+ {
+ "epoch": 1.4209538492475784,
+ "grad_norm": 0.18959026038646698,
+ "learning_rate": 4.345174262734585e-06,
+ "loss": 0.4921241760253906,
+ "step": 5300
+ },
+ {
+ "epoch": 1.4263163186647452,
+ "grad_norm": 0.18584316968917847,
+ "learning_rate": 4.304959785522788e-06,
+ "loss": 0.4892130374908447,
+ "step": 5320
+ },
+ {
+ "epoch": 1.4316787880819117,
+ "grad_norm": 0.17588038742542267,
+ "learning_rate": 4.264745308310992e-06,
+ "loss": 0.4822041988372803,
+ "step": 5340
+ },
+ {
+ "epoch": 1.4370412574990783,
+ "grad_norm": 0.18146033585071564,
+ "learning_rate": 4.224530831099196e-06,
+ "loss": 0.5084807395935058,
+ "step": 5360
+ },
+ {
+ "epoch": 1.4424037269162449,
+ "grad_norm": 0.2251797467470169,
+ "learning_rate": 4.184316353887399e-06,
+ "loss": 0.5146170139312745,
+ "step": 5380
+ },
+ {
+ "epoch": 1.4477661963334114,
+ "grad_norm": 0.18744796514511108,
+ "learning_rate": 4.144101876675603e-06,
+ "loss": 0.5189927577972412,
+ "step": 5400
+ },
+ {
+ "epoch": 1.4531286657505782,
+ "grad_norm": 0.25737133622169495,
+ "learning_rate": 4.103887399463807e-06,
+ "loss": 0.4891658782958984,
+ "step": 5420
+ },
+ {
+ "epoch": 1.4584911351677448,
+ "grad_norm": 0.20580479502677917,
+ "learning_rate": 4.0636729222520105e-06,
+ "loss": 0.4953591823577881,
+ "step": 5440
+ },
+ {
+ "epoch": 1.4638536045849113,
+ "grad_norm": 0.2351546287536621,
+ "learning_rate": 4.023458445040214e-06,
+ "loss": 0.5025320053100586,
+ "step": 5460
+ },
+ {
+ "epoch": 1.4692160740020779,
+ "grad_norm": 0.1819481998682022,
+ "learning_rate": 3.983243967828418e-06,
+ "loss": 0.47151756286621094,
+ "step": 5480
+ },
+ {
+ "epoch": 1.4745785434192444,
+ "grad_norm": 0.20772472023963928,
+ "learning_rate": 3.943029490616622e-06,
+ "loss": 0.4678915023803711,
+ "step": 5500
+ },
+ {
+ "epoch": 1.4799410128364112,
+ "grad_norm": 0.2203037440776825,
+ "learning_rate": 3.902815013404825e-06,
+ "loss": 0.46007452011108396,
+ "step": 5520
+ },
+ {
+ "epoch": 1.4853034822535778,
+ "grad_norm": 0.15371400117874146,
+ "learning_rate": 3.86260053619303e-06,
+ "loss": 0.44407024383544924,
+ "step": 5540
+ },
+ {
+ "epoch": 1.4906659516707443,
+ "grad_norm": 0.2276080846786499,
+ "learning_rate": 3.8223860589812335e-06,
+ "loss": 0.4730556488037109,
+ "step": 5560
+ },
+ {
+ "epoch": 1.4960284210879111,
+ "grad_norm": 0.24482466280460358,
+ "learning_rate": 3.7821715817694376e-06,
+ "loss": 0.5073911666870117,
+ "step": 5580
+ },
+ {
+ "epoch": 1.5013908905050775,
+ "grad_norm": 0.20438458025455475,
+ "learning_rate": 3.741957104557641e-06,
+ "loss": 0.46701641082763673,
+ "step": 5600
+ },
+ {
+ "epoch": 1.5067533599222442,
+ "grad_norm": 0.19854313135147095,
+ "learning_rate": 3.7017426273458446e-06,
+ "loss": 0.46309399604797363,
+ "step": 5620
+ },
+ {
+ "epoch": 1.5121158293394108,
+ "grad_norm": 0.18356069922447205,
+ "learning_rate": 3.6615281501340483e-06,
+ "loss": 0.503613805770874,
+ "step": 5640
+ },
+ {
+ "epoch": 1.5174782987565774,
+ "grad_norm": 0.2009744495153427,
+ "learning_rate": 3.621313672922252e-06,
+ "loss": 0.4765054225921631,
+ "step": 5660
+ },
+ {
+ "epoch": 1.5228407681737441,
+ "grad_norm": 0.3058745563030243,
+ "learning_rate": 3.5810991957104557e-06,
+ "loss": 0.5179148197174073,
+ "step": 5680
+ },
+ {
+ "epoch": 1.5282032375909105,
+ "grad_norm": 0.17671597003936768,
+ "learning_rate": 3.54088471849866e-06,
+ "loss": 0.45907344818115237,
+ "step": 5700
+ },
+ {
+ "epoch": 1.5335657070080773,
+ "grad_norm": 0.22209160029888153,
+ "learning_rate": 3.5006702412868635e-06,
+ "loss": 0.49304862022399903,
+ "step": 5720
+ },
+ {
+ "epoch": 1.5389281764252438,
+ "grad_norm": 0.21018914878368378,
+ "learning_rate": 3.4604557640750672e-06,
+ "loss": 0.5536758422851562,
+ "step": 5740
+ },
+ {
+ "epoch": 1.5442906458424104,
+ "grad_norm": 0.14339996874332428,
+ "learning_rate": 3.420241286863271e-06,
+ "loss": 0.48726091384887693,
+ "step": 5760
+ },
+ {
+ "epoch": 1.5496531152595772,
+ "grad_norm": 0.11419746279716492,
+ "learning_rate": 3.3800268096514746e-06,
+ "loss": 0.4514151573181152,
+ "step": 5780
+ },
+ {
+ "epoch": 1.5550155846767435,
+ "grad_norm": 0.18168962001800537,
+ "learning_rate": 3.3398123324396783e-06,
+ "loss": 0.5279990196228027,
+ "step": 5800
+ },
+ {
+ "epoch": 1.5603780540939103,
+ "grad_norm": 0.24244488775730133,
+ "learning_rate": 3.299597855227882e-06,
+ "loss": 0.49297361373901366,
+ "step": 5820
+ },
+ {
+ "epoch": 1.5657405235110768,
+ "grad_norm": 0.2017296999692917,
+ "learning_rate": 3.2593833780160857e-06,
+ "loss": 0.49305019378662107,
+ "step": 5840
+ },
+ {
+ "epoch": 1.5711029929282434,
+ "grad_norm": 0.22592377662658691,
+ "learning_rate": 3.2191689008042894e-06,
+ "loss": 0.4862989902496338,
+ "step": 5860
+ },
+ {
+ "epoch": 1.5764654623454102,
+ "grad_norm": 0.24772357940673828,
+ "learning_rate": 3.1789544235924935e-06,
+ "loss": 0.45182647705078127,
+ "step": 5880
+ },
+ {
+ "epoch": 1.5818279317625765,
+ "grad_norm": 0.20607218146324158,
+ "learning_rate": 3.1387399463806972e-06,
+ "loss": 0.48905248641967775,
+ "step": 5900
+ },
+ {
+ "epoch": 1.5871904011797433,
+ "grad_norm": 0.1931353509426117,
+ "learning_rate": 3.098525469168901e-06,
+ "loss": 0.5307461261749268,
+ "step": 5920
+ },
+ {
+ "epoch": 1.5925528705969099,
+ "grad_norm": 0.16020581126213074,
+ "learning_rate": 3.0583109919571046e-06,
+ "loss": 0.4672811985015869,
+ "step": 5940
+ },
+ {
+ "epoch": 1.5979153400140764,
+ "grad_norm": 0.23668015003204346,
+ "learning_rate": 3.0180965147453083e-06,
+ "loss": 0.5272688865661621,
+ "step": 5960
+ },
+ {
+ "epoch": 1.6032778094312432,
+ "grad_norm": 0.1916576772928238,
+ "learning_rate": 2.977882037533512e-06,
+ "loss": 0.4859332084655762,
+ "step": 5980
+ },
+ {
+ "epoch": 1.6086402788484095,
+ "grad_norm": 0.23635101318359375,
+ "learning_rate": 2.9376675603217157e-06,
+ "loss": 0.5418910980224609,
+ "step": 6000
+ },
+ {
+ "epoch": 1.6140027482655763,
+ "grad_norm": 0.2404562532901764,
+ "learning_rate": 2.89745308310992e-06,
+ "loss": 0.5449445247650146,
+ "step": 6020
+ },
+ {
+ "epoch": 1.6193652176827429,
+ "grad_norm": 0.20147347450256348,
+ "learning_rate": 2.8572386058981235e-06,
+ "loss": 0.4737790584564209,
+ "step": 6040
+ },
+ {
+ "epoch": 1.6247276870999094,
+ "grad_norm": 0.2455863654613495,
+ "learning_rate": 2.8170241286863272e-06,
+ "loss": 0.4722298145294189,
+ "step": 6060
+ },
+ {
+ "epoch": 1.6300901565170762,
+ "grad_norm": 0.22172148525714874,
+ "learning_rate": 2.776809651474531e-06,
+ "loss": 0.5120372295379638,
+ "step": 6080
+ },
+ {
+ "epoch": 1.6354526259342426,
+ "grad_norm": 0.3848462700843811,
+ "learning_rate": 2.7365951742627346e-06,
+ "loss": 0.5152206897735596,
+ "step": 6100
+ },
+ {
+ "epoch": 1.6408150953514093,
+ "grad_norm": 0.19071047008037567,
+ "learning_rate": 2.6963806970509383e-06,
+ "loss": 0.4757692813873291,
+ "step": 6120
+ },
+ {
+ "epoch": 1.646177564768576,
+ "grad_norm": 0.20568661391735077,
+ "learning_rate": 2.656166219839142e-06,
+ "loss": 0.475917387008667,
+ "step": 6140
+ },
+ {
+ "epoch": 1.6515400341857425,
+ "grad_norm": 0.11777322739362717,
+ "learning_rate": 2.6159517426273457e-06,
+ "loss": 0.5161296367645264,
+ "step": 6160
+ },
+ {
+ "epoch": 1.6569025036029092,
+ "grad_norm": 0.1700555831193924,
+ "learning_rate": 2.5757372654155494e-06,
+ "loss": 0.4715432167053223,
+ "step": 6180
+ },
+ {
+ "epoch": 1.6622649730200756,
+ "grad_norm": 0.18927083909511566,
+ "learning_rate": 2.5355227882037535e-06,
+ "loss": 0.49937710762023924,
+ "step": 6200
+ },
+ {
+ "epoch": 1.6676274424372424,
+ "grad_norm": 0.22097784280776978,
+ "learning_rate": 2.4953083109919572e-06,
+ "loss": 0.43366107940673826,
+ "step": 6220
+ },
+ {
+ "epoch": 1.672989911854409,
+ "grad_norm": 0.2299281805753708,
+ "learning_rate": 2.455093833780161e-06,
+ "loss": 0.5145821094512939,
+ "step": 6240
+ },
+ {
+ "epoch": 1.6783523812715755,
+ "grad_norm": 0.2384844720363617,
+ "learning_rate": 2.4148793565683646e-06,
+ "loss": 0.459308385848999,
+ "step": 6260
+ },
+ {
+ "epoch": 1.6837148506887423,
+ "grad_norm": 0.24471035599708557,
+ "learning_rate": 2.3746648793565683e-06,
+ "loss": 0.4676504611968994,
+ "step": 6280
+ },
+ {
+ "epoch": 1.6890773201059086,
+ "grad_norm": 0.24419866502285004,
+ "learning_rate": 2.334450402144772e-06,
+ "loss": 0.4745138168334961,
+ "step": 6300
+ },
+ {
+ "epoch": 1.6944397895230754,
+ "grad_norm": 0.15896575152873993,
+ "learning_rate": 2.294235924932976e-06,
+ "loss": 0.5073649883270264,
+ "step": 6320
+ },
+ {
+ "epoch": 1.699802258940242,
+ "grad_norm": 0.26504868268966675,
+ "learning_rate": 2.25402144772118e-06,
+ "loss": 0.4534353733062744,
+ "step": 6340
+ },
+ {
+ "epoch": 1.7051647283574085,
+ "grad_norm": 0.2461850792169571,
+ "learning_rate": 2.2138069705093836e-06,
+ "loss": 0.4862947940826416,
+ "step": 6360
+ },
+ {
+ "epoch": 1.7105271977745753,
+ "grad_norm": 0.17332817614078522,
+ "learning_rate": 2.1735924932975873e-06,
+ "loss": 0.5049370765686035,
+ "step": 6380
+ },
+ {
+ "epoch": 1.7158896671917419,
+ "grad_norm": 0.19762548804283142,
+ "learning_rate": 2.133378016085791e-06,
+ "loss": 0.5272616386413574,
+ "step": 6400
+ },
+ {
+ "epoch": 1.7212521366089084,
+ "grad_norm": 0.23265399038791656,
+ "learning_rate": 2.0931635388739946e-06,
+ "loss": 0.47600841522216797,
+ "step": 6420
+ },
+ {
+ "epoch": 1.726614606026075,
+ "grad_norm": 0.20868578553199768,
+ "learning_rate": 2.0529490616621983e-06,
+ "loss": 0.5027226448059082,
+ "step": 6440
+ },
+ {
+ "epoch": 1.7319770754432415,
+ "grad_norm": 0.2851981520652771,
+ "learning_rate": 2.012734584450402e-06,
+ "loss": 0.5288124561309815,
+ "step": 6460
+ },
+ {
+ "epoch": 1.7373395448604083,
+ "grad_norm": 0.20086587965488434,
+ "learning_rate": 1.9725201072386057e-06,
+ "loss": 0.4625516891479492,
+ "step": 6480
+ },
+ {
+ "epoch": 1.7427020142775749,
+ "grad_norm": 0.24060192704200745,
+ "learning_rate": 1.93230563002681e-06,
+ "loss": 0.4843903541564941,
+ "step": 6500
+ },
+ {
+ "epoch": 1.7480644836947414,
+ "grad_norm": 0.33561915159225464,
+ "learning_rate": 1.8920911528150133e-06,
+ "loss": 0.4823720932006836,
+ "step": 6520
+ },
+ {
+ "epoch": 1.753426953111908,
+ "grad_norm": 0.2510465383529663,
+ "learning_rate": 1.851876675603217e-06,
+ "loss": 0.46517143249511717,
+ "step": 6540
+ },
+ {
+ "epoch": 1.7587894225290746,
+ "grad_norm": 0.2631177604198456,
+ "learning_rate": 1.811662198391421e-06,
+ "loss": 0.5004732131958007,
+ "step": 6560
+ },
+ {
+ "epoch": 1.7641518919462413,
+ "grad_norm": 0.3493230640888214,
+ "learning_rate": 1.7714477211796249e-06,
+ "loss": 0.523811674118042,
+ "step": 6580
+ },
+ {
+ "epoch": 1.769514361363408,
+ "grad_norm": 0.1742691546678543,
+ "learning_rate": 1.7312332439678286e-06,
+ "loss": 0.5276295661926269,
+ "step": 6600
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 8.127024591687352e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-6600/training_args.bin b/checkpoint-6600/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-6600/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-6800/README.md b/checkpoint-6800/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-6800/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-6800/adapter_config.json b/checkpoint-6800/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-6800/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-6800/adapter_model.safetensors b/checkpoint-6800/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..4e5ced89a23e0ee4d928e82115bef05251adaca4
--- /dev/null
+++ b/checkpoint-6800/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ed2991897b87a6505a11f1e261458d003853a423454d0c695a836279c91dedb0
+size 50360752
diff --git a/checkpoint-6800/chat_template.jinja b/checkpoint-6800/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-6800/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-6800/optimizer.pt b/checkpoint-6800/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..6f6ead2116d20e12680868a130aa5f4c5ea5b44a
--- /dev/null
+++ b/checkpoint-6800/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b06c5d417d484fe30a3f5e545864a81855d53a261a19a9221db0ec0ba0ebdf6e
+size 100828235
diff --git a/checkpoint-6800/rng_state.pth b/checkpoint-6800/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..838e7749ecb01948a0070debc676a56f542d313f
--- /dev/null
+++ b/checkpoint-6800/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:289bb67593ed6c88781bd4c977b0573cb2a3f9659d6a230efdd16085a8520afe
+size 14645
diff --git a/checkpoint-6800/scheduler.pt b/checkpoint-6800/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..b279a01beb2afedbb029a4aa724d9541f3112902
--- /dev/null
+++ b/checkpoint-6800/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:88a909cddd57590705e606f63cdba0fe6697d4a9729afb4be3dfced9292a5a67
+size 1465
diff --git a/checkpoint-6800/tokenizer.json b/checkpoint-6800/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-6800/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-6800/tokenizer_config.json b/checkpoint-6800/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-6800/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-6800/trainer_state.json b/checkpoint-6800/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..9b1cc6607083179f9687b045a191c304affafc26
--- /dev/null
+++ b/checkpoint-6800/trainer_state.json
@@ -0,0 +1,2414 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.823139055535074,
+ "eval_steps": 500,
+ "global_step": 6800,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ },
+ {
+ "epoch": 1.185005194892248,
+ "grad_norm": 0.16667844355106354,
+ "learning_rate": 6.114611260053619e-06,
+ "loss": 0.5159809589385986,
+ "step": 4420
+ },
+ {
+ "epoch": 1.1903676643094145,
+ "grad_norm": 0.19206570088863373,
+ "learning_rate": 6.074396782841823e-06,
+ "loss": 0.47541089057922364,
+ "step": 4440
+ },
+ {
+ "epoch": 1.195730133726581,
+ "grad_norm": 0.17394617199897766,
+ "learning_rate": 6.034182305630027e-06,
+ "loss": 0.5470661640167236,
+ "step": 4460
+ },
+ {
+ "epoch": 1.2010926031437477,
+ "grad_norm": 0.210404634475708,
+ "learning_rate": 5.993967828418231e-06,
+ "loss": 0.5377882957458496,
+ "step": 4480
+ },
+ {
+ "epoch": 1.2064550725609142,
+ "grad_norm": 0.18084648251533508,
+ "learning_rate": 5.953753351206435e-06,
+ "loss": 0.5037185192108155,
+ "step": 4500
+ },
+ {
+ "epoch": 1.211817541978081,
+ "grad_norm": 0.23707027733325958,
+ "learning_rate": 5.913538873994639e-06,
+ "loss": 0.4822190284729004,
+ "step": 4520
+ },
+ {
+ "epoch": 1.2171800113952476,
+ "grad_norm": 0.16474473476409912,
+ "learning_rate": 5.873324396782842e-06,
+ "loss": 0.46645288467407225,
+ "step": 4540
+ },
+ {
+ "epoch": 1.2225424808124141,
+ "grad_norm": 0.2142348438501358,
+ "learning_rate": 5.833109919571046e-06,
+ "loss": 0.5255855560302735,
+ "step": 4560
+ },
+ {
+ "epoch": 1.2279049502295807,
+ "grad_norm": 0.2531765103340149,
+ "learning_rate": 5.79289544235925e-06,
+ "loss": 0.507044792175293,
+ "step": 4580
+ },
+ {
+ "epoch": 1.2332674196467472,
+ "grad_norm": 0.2553550899028778,
+ "learning_rate": 5.7526809651474535e-06,
+ "loss": 0.4767824649810791,
+ "step": 4600
+ },
+ {
+ "epoch": 1.238629889063914,
+ "grad_norm": 0.14484412968158722,
+ "learning_rate": 5.712466487935657e-06,
+ "loss": 0.4675601005554199,
+ "step": 4620
+ },
+ {
+ "epoch": 1.2439923584810806,
+ "grad_norm": 0.14328251779079437,
+ "learning_rate": 5.672252010723861e-06,
+ "loss": 0.4956005573272705,
+ "step": 4640
+ },
+ {
+ "epoch": 1.2493548278982471,
+ "grad_norm": 0.1739245355129242,
+ "learning_rate": 5.632037533512065e-06,
+ "loss": 0.48583345413208007,
+ "step": 4660
+ },
+ {
+ "epoch": 1.2547172973154137,
+ "grad_norm": 0.21294184029102325,
+ "learning_rate": 5.591823056300268e-06,
+ "loss": 0.520921277999878,
+ "step": 4680
+ },
+ {
+ "epoch": 1.2600797667325803,
+ "grad_norm": 0.25132355093955994,
+ "learning_rate": 5.551608579088472e-06,
+ "loss": 0.5295385837554931,
+ "step": 4700
+ },
+ {
+ "epoch": 1.265442236149747,
+ "grad_norm": 0.18603841960430145,
+ "learning_rate": 5.511394101876676e-06,
+ "loss": 0.47570199966430665,
+ "step": 4720
+ },
+ {
+ "epoch": 1.2708047055669136,
+ "grad_norm": 0.19883134961128235,
+ "learning_rate": 5.471179624664879e-06,
+ "loss": 0.5016080379486084,
+ "step": 4740
+ },
+ {
+ "epoch": 1.2761671749840802,
+ "grad_norm": 0.19640181958675385,
+ "learning_rate": 5.430965147453083e-06,
+ "loss": 0.4999081134796143,
+ "step": 4760
+ },
+ {
+ "epoch": 1.2815296444012467,
+ "grad_norm": 0.2584764361381531,
+ "learning_rate": 5.390750670241287e-06,
+ "loss": 0.4780082702636719,
+ "step": 4780
+ },
+ {
+ "epoch": 1.2868921138184133,
+ "grad_norm": 0.2925741374492645,
+ "learning_rate": 5.3505361930294905e-06,
+ "loss": 0.5131395816802978,
+ "step": 4800
+ },
+ {
+ "epoch": 1.29225458323558,
+ "grad_norm": 0.18971531093120575,
+ "learning_rate": 5.310321715817694e-06,
+ "loss": 0.455674409866333,
+ "step": 4820
+ },
+ {
+ "epoch": 1.2976170526527466,
+ "grad_norm": 0.16778405010700226,
+ "learning_rate": 5.270107238605898e-06,
+ "loss": 0.5070962905883789,
+ "step": 4840
+ },
+ {
+ "epoch": 1.3029795220699132,
+ "grad_norm": 0.30026957392692566,
+ "learning_rate": 5.2298927613941016e-06,
+ "loss": 0.5120027542114258,
+ "step": 4860
+ },
+ {
+ "epoch": 1.3083419914870797,
+ "grad_norm": 0.17846634984016418,
+ "learning_rate": 5.189678284182305e-06,
+ "loss": 0.5114477157592774,
+ "step": 4880
+ },
+ {
+ "epoch": 1.3137044609042463,
+ "grad_norm": 0.1962418258190155,
+ "learning_rate": 5.149463806970509e-06,
+ "loss": 0.5043613910675049,
+ "step": 4900
+ },
+ {
+ "epoch": 1.319066930321413,
+ "grad_norm": 0.18446756899356842,
+ "learning_rate": 5.1092493297587135e-06,
+ "loss": 0.5396455287933349,
+ "step": 4920
+ },
+ {
+ "epoch": 1.3244293997385796,
+ "grad_norm": 0.20886844396591187,
+ "learning_rate": 5.069034852546917e-06,
+ "loss": 0.4879767417907715,
+ "step": 4940
+ },
+ {
+ "epoch": 1.3297918691557462,
+ "grad_norm": 0.16687901318073273,
+ "learning_rate": 5.028820375335121e-06,
+ "loss": 0.5014327049255372,
+ "step": 4960
+ },
+ {
+ "epoch": 1.3351543385729128,
+ "grad_norm": 0.19595153629779816,
+ "learning_rate": 4.988605898123325e-06,
+ "loss": 0.5375277996063232,
+ "step": 4980
+ },
+ {
+ "epoch": 1.3405168079900793,
+ "grad_norm": 0.2372344732284546,
+ "learning_rate": 4.948391420911528e-06,
+ "loss": 0.5020076274871826,
+ "step": 5000
+ },
+ {
+ "epoch": 1.345879277407246,
+ "grad_norm": 0.21030014753341675,
+ "learning_rate": 4.908176943699732e-06,
+ "loss": 0.5111066818237304,
+ "step": 5020
+ },
+ {
+ "epoch": 1.3512417468244127,
+ "grad_norm": 0.1866692751646042,
+ "learning_rate": 4.867962466487936e-06,
+ "loss": 0.4515383720397949,
+ "step": 5040
+ },
+ {
+ "epoch": 1.3566042162415792,
+ "grad_norm": 0.22531798481941223,
+ "learning_rate": 4.827747989276139e-06,
+ "loss": 0.4757690906524658,
+ "step": 5060
+ },
+ {
+ "epoch": 1.3619666856587458,
+ "grad_norm": 0.15868768095970154,
+ "learning_rate": 4.787533512064343e-06,
+ "loss": 0.45842318534851073,
+ "step": 5080
+ },
+ {
+ "epoch": 1.3673291550759124,
+ "grad_norm": 0.24528546631336212,
+ "learning_rate": 4.747319034852547e-06,
+ "loss": 0.47269258499145506,
+ "step": 5100
+ },
+ {
+ "epoch": 1.3726916244930791,
+ "grad_norm": 0.17387732863426208,
+ "learning_rate": 4.707104557640751e-06,
+ "loss": 0.5103805065155029,
+ "step": 5120
+ },
+ {
+ "epoch": 1.3780540939102457,
+ "grad_norm": 0.20686905086040497,
+ "learning_rate": 4.666890080428955e-06,
+ "loss": 0.5135180950164795,
+ "step": 5140
+ },
+ {
+ "epoch": 1.3834165633274123,
+ "grad_norm": 0.19599783420562744,
+ "learning_rate": 4.626675603217159e-06,
+ "loss": 0.5045839786529541,
+ "step": 5160
+ },
+ {
+ "epoch": 1.3887790327445788,
+ "grad_norm": 0.2585010528564453,
+ "learning_rate": 4.586461126005362e-06,
+ "loss": 0.45903496742248534,
+ "step": 5180
+ },
+ {
+ "epoch": 1.3941415021617454,
+ "grad_norm": 0.1688319593667984,
+ "learning_rate": 4.546246648793566e-06,
+ "loss": 0.5017509937286377,
+ "step": 5200
+ },
+ {
+ "epoch": 1.3995039715789122,
+ "grad_norm": 0.21520815789699554,
+ "learning_rate": 4.50603217158177e-06,
+ "loss": 0.48459539413452146,
+ "step": 5220
+ },
+ {
+ "epoch": 1.4048664409960787,
+ "grad_norm": 0.20514647662639618,
+ "learning_rate": 4.4658176943699735e-06,
+ "loss": 0.5073423862457276,
+ "step": 5240
+ },
+ {
+ "epoch": 1.4102289104132453,
+ "grad_norm": 0.21835413575172424,
+ "learning_rate": 4.425603217158177e-06,
+ "loss": 0.5290310382843018,
+ "step": 5260
+ },
+ {
+ "epoch": 1.4155913798304118,
+ "grad_norm": 0.28042587637901306,
+ "learning_rate": 4.385388739946381e-06,
+ "loss": 0.4823312759399414,
+ "step": 5280
+ },
+ {
+ "epoch": 1.4209538492475784,
+ "grad_norm": 0.18959026038646698,
+ "learning_rate": 4.345174262734585e-06,
+ "loss": 0.4921241760253906,
+ "step": 5300
+ },
+ {
+ "epoch": 1.4263163186647452,
+ "grad_norm": 0.18584316968917847,
+ "learning_rate": 4.304959785522788e-06,
+ "loss": 0.4892130374908447,
+ "step": 5320
+ },
+ {
+ "epoch": 1.4316787880819117,
+ "grad_norm": 0.17588038742542267,
+ "learning_rate": 4.264745308310992e-06,
+ "loss": 0.4822041988372803,
+ "step": 5340
+ },
+ {
+ "epoch": 1.4370412574990783,
+ "grad_norm": 0.18146033585071564,
+ "learning_rate": 4.224530831099196e-06,
+ "loss": 0.5084807395935058,
+ "step": 5360
+ },
+ {
+ "epoch": 1.4424037269162449,
+ "grad_norm": 0.2251797467470169,
+ "learning_rate": 4.184316353887399e-06,
+ "loss": 0.5146170139312745,
+ "step": 5380
+ },
+ {
+ "epoch": 1.4477661963334114,
+ "grad_norm": 0.18744796514511108,
+ "learning_rate": 4.144101876675603e-06,
+ "loss": 0.5189927577972412,
+ "step": 5400
+ },
+ {
+ "epoch": 1.4531286657505782,
+ "grad_norm": 0.25737133622169495,
+ "learning_rate": 4.103887399463807e-06,
+ "loss": 0.4891658782958984,
+ "step": 5420
+ },
+ {
+ "epoch": 1.4584911351677448,
+ "grad_norm": 0.20580479502677917,
+ "learning_rate": 4.0636729222520105e-06,
+ "loss": 0.4953591823577881,
+ "step": 5440
+ },
+ {
+ "epoch": 1.4638536045849113,
+ "grad_norm": 0.2351546287536621,
+ "learning_rate": 4.023458445040214e-06,
+ "loss": 0.5025320053100586,
+ "step": 5460
+ },
+ {
+ "epoch": 1.4692160740020779,
+ "grad_norm": 0.1819481998682022,
+ "learning_rate": 3.983243967828418e-06,
+ "loss": 0.47151756286621094,
+ "step": 5480
+ },
+ {
+ "epoch": 1.4745785434192444,
+ "grad_norm": 0.20772472023963928,
+ "learning_rate": 3.943029490616622e-06,
+ "loss": 0.4678915023803711,
+ "step": 5500
+ },
+ {
+ "epoch": 1.4799410128364112,
+ "grad_norm": 0.2203037440776825,
+ "learning_rate": 3.902815013404825e-06,
+ "loss": 0.46007452011108396,
+ "step": 5520
+ },
+ {
+ "epoch": 1.4853034822535778,
+ "grad_norm": 0.15371400117874146,
+ "learning_rate": 3.86260053619303e-06,
+ "loss": 0.44407024383544924,
+ "step": 5540
+ },
+ {
+ "epoch": 1.4906659516707443,
+ "grad_norm": 0.2276080846786499,
+ "learning_rate": 3.8223860589812335e-06,
+ "loss": 0.4730556488037109,
+ "step": 5560
+ },
+ {
+ "epoch": 1.4960284210879111,
+ "grad_norm": 0.24482466280460358,
+ "learning_rate": 3.7821715817694376e-06,
+ "loss": 0.5073911666870117,
+ "step": 5580
+ },
+ {
+ "epoch": 1.5013908905050775,
+ "grad_norm": 0.20438458025455475,
+ "learning_rate": 3.741957104557641e-06,
+ "loss": 0.46701641082763673,
+ "step": 5600
+ },
+ {
+ "epoch": 1.5067533599222442,
+ "grad_norm": 0.19854313135147095,
+ "learning_rate": 3.7017426273458446e-06,
+ "loss": 0.46309399604797363,
+ "step": 5620
+ },
+ {
+ "epoch": 1.5121158293394108,
+ "grad_norm": 0.18356069922447205,
+ "learning_rate": 3.6615281501340483e-06,
+ "loss": 0.503613805770874,
+ "step": 5640
+ },
+ {
+ "epoch": 1.5174782987565774,
+ "grad_norm": 0.2009744495153427,
+ "learning_rate": 3.621313672922252e-06,
+ "loss": 0.4765054225921631,
+ "step": 5660
+ },
+ {
+ "epoch": 1.5228407681737441,
+ "grad_norm": 0.3058745563030243,
+ "learning_rate": 3.5810991957104557e-06,
+ "loss": 0.5179148197174073,
+ "step": 5680
+ },
+ {
+ "epoch": 1.5282032375909105,
+ "grad_norm": 0.17671597003936768,
+ "learning_rate": 3.54088471849866e-06,
+ "loss": 0.45907344818115237,
+ "step": 5700
+ },
+ {
+ "epoch": 1.5335657070080773,
+ "grad_norm": 0.22209160029888153,
+ "learning_rate": 3.5006702412868635e-06,
+ "loss": 0.49304862022399903,
+ "step": 5720
+ },
+ {
+ "epoch": 1.5389281764252438,
+ "grad_norm": 0.21018914878368378,
+ "learning_rate": 3.4604557640750672e-06,
+ "loss": 0.5536758422851562,
+ "step": 5740
+ },
+ {
+ "epoch": 1.5442906458424104,
+ "grad_norm": 0.14339996874332428,
+ "learning_rate": 3.420241286863271e-06,
+ "loss": 0.48726091384887693,
+ "step": 5760
+ },
+ {
+ "epoch": 1.5496531152595772,
+ "grad_norm": 0.11419746279716492,
+ "learning_rate": 3.3800268096514746e-06,
+ "loss": 0.4514151573181152,
+ "step": 5780
+ },
+ {
+ "epoch": 1.5550155846767435,
+ "grad_norm": 0.18168962001800537,
+ "learning_rate": 3.3398123324396783e-06,
+ "loss": 0.5279990196228027,
+ "step": 5800
+ },
+ {
+ "epoch": 1.5603780540939103,
+ "grad_norm": 0.24244488775730133,
+ "learning_rate": 3.299597855227882e-06,
+ "loss": 0.49297361373901366,
+ "step": 5820
+ },
+ {
+ "epoch": 1.5657405235110768,
+ "grad_norm": 0.2017296999692917,
+ "learning_rate": 3.2593833780160857e-06,
+ "loss": 0.49305019378662107,
+ "step": 5840
+ },
+ {
+ "epoch": 1.5711029929282434,
+ "grad_norm": 0.22592377662658691,
+ "learning_rate": 3.2191689008042894e-06,
+ "loss": 0.4862989902496338,
+ "step": 5860
+ },
+ {
+ "epoch": 1.5764654623454102,
+ "grad_norm": 0.24772357940673828,
+ "learning_rate": 3.1789544235924935e-06,
+ "loss": 0.45182647705078127,
+ "step": 5880
+ },
+ {
+ "epoch": 1.5818279317625765,
+ "grad_norm": 0.20607218146324158,
+ "learning_rate": 3.1387399463806972e-06,
+ "loss": 0.48905248641967775,
+ "step": 5900
+ },
+ {
+ "epoch": 1.5871904011797433,
+ "grad_norm": 0.1931353509426117,
+ "learning_rate": 3.098525469168901e-06,
+ "loss": 0.5307461261749268,
+ "step": 5920
+ },
+ {
+ "epoch": 1.5925528705969099,
+ "grad_norm": 0.16020581126213074,
+ "learning_rate": 3.0583109919571046e-06,
+ "loss": 0.4672811985015869,
+ "step": 5940
+ },
+ {
+ "epoch": 1.5979153400140764,
+ "grad_norm": 0.23668015003204346,
+ "learning_rate": 3.0180965147453083e-06,
+ "loss": 0.5272688865661621,
+ "step": 5960
+ },
+ {
+ "epoch": 1.6032778094312432,
+ "grad_norm": 0.1916576772928238,
+ "learning_rate": 2.977882037533512e-06,
+ "loss": 0.4859332084655762,
+ "step": 5980
+ },
+ {
+ "epoch": 1.6086402788484095,
+ "grad_norm": 0.23635101318359375,
+ "learning_rate": 2.9376675603217157e-06,
+ "loss": 0.5418910980224609,
+ "step": 6000
+ },
+ {
+ "epoch": 1.6140027482655763,
+ "grad_norm": 0.2404562532901764,
+ "learning_rate": 2.89745308310992e-06,
+ "loss": 0.5449445247650146,
+ "step": 6020
+ },
+ {
+ "epoch": 1.6193652176827429,
+ "grad_norm": 0.20147347450256348,
+ "learning_rate": 2.8572386058981235e-06,
+ "loss": 0.4737790584564209,
+ "step": 6040
+ },
+ {
+ "epoch": 1.6247276870999094,
+ "grad_norm": 0.2455863654613495,
+ "learning_rate": 2.8170241286863272e-06,
+ "loss": 0.4722298145294189,
+ "step": 6060
+ },
+ {
+ "epoch": 1.6300901565170762,
+ "grad_norm": 0.22172148525714874,
+ "learning_rate": 2.776809651474531e-06,
+ "loss": 0.5120372295379638,
+ "step": 6080
+ },
+ {
+ "epoch": 1.6354526259342426,
+ "grad_norm": 0.3848462700843811,
+ "learning_rate": 2.7365951742627346e-06,
+ "loss": 0.5152206897735596,
+ "step": 6100
+ },
+ {
+ "epoch": 1.6408150953514093,
+ "grad_norm": 0.19071047008037567,
+ "learning_rate": 2.6963806970509383e-06,
+ "loss": 0.4757692813873291,
+ "step": 6120
+ },
+ {
+ "epoch": 1.646177564768576,
+ "grad_norm": 0.20568661391735077,
+ "learning_rate": 2.656166219839142e-06,
+ "loss": 0.475917387008667,
+ "step": 6140
+ },
+ {
+ "epoch": 1.6515400341857425,
+ "grad_norm": 0.11777322739362717,
+ "learning_rate": 2.6159517426273457e-06,
+ "loss": 0.5161296367645264,
+ "step": 6160
+ },
+ {
+ "epoch": 1.6569025036029092,
+ "grad_norm": 0.1700555831193924,
+ "learning_rate": 2.5757372654155494e-06,
+ "loss": 0.4715432167053223,
+ "step": 6180
+ },
+ {
+ "epoch": 1.6622649730200756,
+ "grad_norm": 0.18927083909511566,
+ "learning_rate": 2.5355227882037535e-06,
+ "loss": 0.49937710762023924,
+ "step": 6200
+ },
+ {
+ "epoch": 1.6676274424372424,
+ "grad_norm": 0.22097784280776978,
+ "learning_rate": 2.4953083109919572e-06,
+ "loss": 0.43366107940673826,
+ "step": 6220
+ },
+ {
+ "epoch": 1.672989911854409,
+ "grad_norm": 0.2299281805753708,
+ "learning_rate": 2.455093833780161e-06,
+ "loss": 0.5145821094512939,
+ "step": 6240
+ },
+ {
+ "epoch": 1.6783523812715755,
+ "grad_norm": 0.2384844720363617,
+ "learning_rate": 2.4148793565683646e-06,
+ "loss": 0.459308385848999,
+ "step": 6260
+ },
+ {
+ "epoch": 1.6837148506887423,
+ "grad_norm": 0.24471035599708557,
+ "learning_rate": 2.3746648793565683e-06,
+ "loss": 0.4676504611968994,
+ "step": 6280
+ },
+ {
+ "epoch": 1.6890773201059086,
+ "grad_norm": 0.24419866502285004,
+ "learning_rate": 2.334450402144772e-06,
+ "loss": 0.4745138168334961,
+ "step": 6300
+ },
+ {
+ "epoch": 1.6944397895230754,
+ "grad_norm": 0.15896575152873993,
+ "learning_rate": 2.294235924932976e-06,
+ "loss": 0.5073649883270264,
+ "step": 6320
+ },
+ {
+ "epoch": 1.699802258940242,
+ "grad_norm": 0.26504868268966675,
+ "learning_rate": 2.25402144772118e-06,
+ "loss": 0.4534353733062744,
+ "step": 6340
+ },
+ {
+ "epoch": 1.7051647283574085,
+ "grad_norm": 0.2461850792169571,
+ "learning_rate": 2.2138069705093836e-06,
+ "loss": 0.4862947940826416,
+ "step": 6360
+ },
+ {
+ "epoch": 1.7105271977745753,
+ "grad_norm": 0.17332817614078522,
+ "learning_rate": 2.1735924932975873e-06,
+ "loss": 0.5049370765686035,
+ "step": 6380
+ },
+ {
+ "epoch": 1.7158896671917419,
+ "grad_norm": 0.19762548804283142,
+ "learning_rate": 2.133378016085791e-06,
+ "loss": 0.5272616386413574,
+ "step": 6400
+ },
+ {
+ "epoch": 1.7212521366089084,
+ "grad_norm": 0.23265399038791656,
+ "learning_rate": 2.0931635388739946e-06,
+ "loss": 0.47600841522216797,
+ "step": 6420
+ },
+ {
+ "epoch": 1.726614606026075,
+ "grad_norm": 0.20868578553199768,
+ "learning_rate": 2.0529490616621983e-06,
+ "loss": 0.5027226448059082,
+ "step": 6440
+ },
+ {
+ "epoch": 1.7319770754432415,
+ "grad_norm": 0.2851981520652771,
+ "learning_rate": 2.012734584450402e-06,
+ "loss": 0.5288124561309815,
+ "step": 6460
+ },
+ {
+ "epoch": 1.7373395448604083,
+ "grad_norm": 0.20086587965488434,
+ "learning_rate": 1.9725201072386057e-06,
+ "loss": 0.4625516891479492,
+ "step": 6480
+ },
+ {
+ "epoch": 1.7427020142775749,
+ "grad_norm": 0.24060192704200745,
+ "learning_rate": 1.93230563002681e-06,
+ "loss": 0.4843903541564941,
+ "step": 6500
+ },
+ {
+ "epoch": 1.7480644836947414,
+ "grad_norm": 0.33561915159225464,
+ "learning_rate": 1.8920911528150133e-06,
+ "loss": 0.4823720932006836,
+ "step": 6520
+ },
+ {
+ "epoch": 1.753426953111908,
+ "grad_norm": 0.2510465383529663,
+ "learning_rate": 1.851876675603217e-06,
+ "loss": 0.46517143249511717,
+ "step": 6540
+ },
+ {
+ "epoch": 1.7587894225290746,
+ "grad_norm": 0.2631177604198456,
+ "learning_rate": 1.811662198391421e-06,
+ "loss": 0.5004732131958007,
+ "step": 6560
+ },
+ {
+ "epoch": 1.7641518919462413,
+ "grad_norm": 0.3493230640888214,
+ "learning_rate": 1.7714477211796249e-06,
+ "loss": 0.523811674118042,
+ "step": 6580
+ },
+ {
+ "epoch": 1.769514361363408,
+ "grad_norm": 0.1742691546678543,
+ "learning_rate": 1.7312332439678286e-06,
+ "loss": 0.5276295661926269,
+ "step": 6600
+ },
+ {
+ "epoch": 1.7748768307805745,
+ "grad_norm": 0.16134823858737946,
+ "learning_rate": 1.6910187667560323e-06,
+ "loss": 0.5352637290954589,
+ "step": 6620
+ },
+ {
+ "epoch": 1.780239300197741,
+ "grad_norm": 0.20977018773555756,
+ "learning_rate": 1.650804289544236e-06,
+ "loss": 0.4955774784088135,
+ "step": 6640
+ },
+ {
+ "epoch": 1.7856017696149076,
+ "grad_norm": 0.20511005818843842,
+ "learning_rate": 1.6105898123324397e-06,
+ "loss": 0.48643174171447756,
+ "step": 6660
+ },
+ {
+ "epoch": 1.7909642390320744,
+ "grad_norm": 0.23870044946670532,
+ "learning_rate": 1.5703753351206434e-06,
+ "loss": 0.4673162460327148,
+ "step": 6680
+ },
+ {
+ "epoch": 1.796326708449241,
+ "grad_norm": 0.21660065650939941,
+ "learning_rate": 1.5301608579088473e-06,
+ "loss": 0.5381903648376465,
+ "step": 6700
+ },
+ {
+ "epoch": 1.8016891778664075,
+ "grad_norm": 0.26977139711380005,
+ "learning_rate": 1.489946380697051e-06,
+ "loss": 0.42094998359680175,
+ "step": 6720
+ },
+ {
+ "epoch": 1.807051647283574,
+ "grad_norm": 0.2088550478219986,
+ "learning_rate": 1.4497319034852549e-06,
+ "loss": 0.49211792945861815,
+ "step": 6740
+ },
+ {
+ "epoch": 1.8124141167007406,
+ "grad_norm": 0.18141885101795197,
+ "learning_rate": 1.4095174262734586e-06,
+ "loss": 0.46572179794311525,
+ "step": 6760
+ },
+ {
+ "epoch": 1.8177765861179074,
+ "grad_norm": 0.2200685739517212,
+ "learning_rate": 1.3693029490616623e-06,
+ "loss": 0.4996177196502686,
+ "step": 6780
+ },
+ {
+ "epoch": 1.823139055535074,
+ "grad_norm": 0.19545452296733856,
+ "learning_rate": 1.329088471849866e-06,
+ "loss": 0.4731945514678955,
+ "step": 6800
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 8.373542625780265e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-6800/training_args.bin b/checkpoint-6800/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-6800/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-7000/README.md b/checkpoint-7000/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-7000/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-7000/adapter_config.json b/checkpoint-7000/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-7000/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-7000/adapter_model.safetensors b/checkpoint-7000/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..5eeaeda2a79ecf256eb58ab7d9efbca0300c5110
--- /dev/null
+++ b/checkpoint-7000/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b4663c3b7fb54ea5bdcd3bc60c04176a5052f0c730e5bd2de51576073db6ce7d
+size 50360752
diff --git a/checkpoint-7000/chat_template.jinja b/checkpoint-7000/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-7000/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-7000/optimizer.pt b/checkpoint-7000/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..cef40d2996fba6f4adb2c9b2294c1750fb72af35
--- /dev/null
+++ b/checkpoint-7000/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d1ef1c8f148b60b0cd7d3d6796071840dec41f4ba0a97495b85fb7c90ffdee79
+size 100828235
diff --git a/checkpoint-7000/rng_state.pth b/checkpoint-7000/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..9dcf635d0af42decbc7ed01d7bf422f16e7c6d6f
--- /dev/null
+++ b/checkpoint-7000/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:71b1ac65a6bc1245f150fab07c74f73f6f644042151c39b0c3ed7f07bddf91af
+size 14645
diff --git a/checkpoint-7000/scheduler.pt b/checkpoint-7000/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..fca355e69583b905245f88112070c3006ad292d2
--- /dev/null
+++ b/checkpoint-7000/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:421f24085050d20209d7d03e1cc26d43deaa27183041a8fc1f7d3f63c3e6666c
+size 1465
diff --git a/checkpoint-7000/tokenizer.json b/checkpoint-7000/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-7000/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-7000/tokenizer_config.json b/checkpoint-7000/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-7000/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-7000/trainer_state.json b/checkpoint-7000/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..23dfec5dec1b6d1b14c0f9bc5e45c9ee5dc992a0
--- /dev/null
+++ b/checkpoint-7000/trainer_state.json
@@ -0,0 +1,2484 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.87676374970674,
+ "eval_steps": 500,
+ "global_step": 7000,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ },
+ {
+ "epoch": 1.185005194892248,
+ "grad_norm": 0.16667844355106354,
+ "learning_rate": 6.114611260053619e-06,
+ "loss": 0.5159809589385986,
+ "step": 4420
+ },
+ {
+ "epoch": 1.1903676643094145,
+ "grad_norm": 0.19206570088863373,
+ "learning_rate": 6.074396782841823e-06,
+ "loss": 0.47541089057922364,
+ "step": 4440
+ },
+ {
+ "epoch": 1.195730133726581,
+ "grad_norm": 0.17394617199897766,
+ "learning_rate": 6.034182305630027e-06,
+ "loss": 0.5470661640167236,
+ "step": 4460
+ },
+ {
+ "epoch": 1.2010926031437477,
+ "grad_norm": 0.210404634475708,
+ "learning_rate": 5.993967828418231e-06,
+ "loss": 0.5377882957458496,
+ "step": 4480
+ },
+ {
+ "epoch": 1.2064550725609142,
+ "grad_norm": 0.18084648251533508,
+ "learning_rate": 5.953753351206435e-06,
+ "loss": 0.5037185192108155,
+ "step": 4500
+ },
+ {
+ "epoch": 1.211817541978081,
+ "grad_norm": 0.23707027733325958,
+ "learning_rate": 5.913538873994639e-06,
+ "loss": 0.4822190284729004,
+ "step": 4520
+ },
+ {
+ "epoch": 1.2171800113952476,
+ "grad_norm": 0.16474473476409912,
+ "learning_rate": 5.873324396782842e-06,
+ "loss": 0.46645288467407225,
+ "step": 4540
+ },
+ {
+ "epoch": 1.2225424808124141,
+ "grad_norm": 0.2142348438501358,
+ "learning_rate": 5.833109919571046e-06,
+ "loss": 0.5255855560302735,
+ "step": 4560
+ },
+ {
+ "epoch": 1.2279049502295807,
+ "grad_norm": 0.2531765103340149,
+ "learning_rate": 5.79289544235925e-06,
+ "loss": 0.507044792175293,
+ "step": 4580
+ },
+ {
+ "epoch": 1.2332674196467472,
+ "grad_norm": 0.2553550899028778,
+ "learning_rate": 5.7526809651474535e-06,
+ "loss": 0.4767824649810791,
+ "step": 4600
+ },
+ {
+ "epoch": 1.238629889063914,
+ "grad_norm": 0.14484412968158722,
+ "learning_rate": 5.712466487935657e-06,
+ "loss": 0.4675601005554199,
+ "step": 4620
+ },
+ {
+ "epoch": 1.2439923584810806,
+ "grad_norm": 0.14328251779079437,
+ "learning_rate": 5.672252010723861e-06,
+ "loss": 0.4956005573272705,
+ "step": 4640
+ },
+ {
+ "epoch": 1.2493548278982471,
+ "grad_norm": 0.1739245355129242,
+ "learning_rate": 5.632037533512065e-06,
+ "loss": 0.48583345413208007,
+ "step": 4660
+ },
+ {
+ "epoch": 1.2547172973154137,
+ "grad_norm": 0.21294184029102325,
+ "learning_rate": 5.591823056300268e-06,
+ "loss": 0.520921277999878,
+ "step": 4680
+ },
+ {
+ "epoch": 1.2600797667325803,
+ "grad_norm": 0.25132355093955994,
+ "learning_rate": 5.551608579088472e-06,
+ "loss": 0.5295385837554931,
+ "step": 4700
+ },
+ {
+ "epoch": 1.265442236149747,
+ "grad_norm": 0.18603841960430145,
+ "learning_rate": 5.511394101876676e-06,
+ "loss": 0.47570199966430665,
+ "step": 4720
+ },
+ {
+ "epoch": 1.2708047055669136,
+ "grad_norm": 0.19883134961128235,
+ "learning_rate": 5.471179624664879e-06,
+ "loss": 0.5016080379486084,
+ "step": 4740
+ },
+ {
+ "epoch": 1.2761671749840802,
+ "grad_norm": 0.19640181958675385,
+ "learning_rate": 5.430965147453083e-06,
+ "loss": 0.4999081134796143,
+ "step": 4760
+ },
+ {
+ "epoch": 1.2815296444012467,
+ "grad_norm": 0.2584764361381531,
+ "learning_rate": 5.390750670241287e-06,
+ "loss": 0.4780082702636719,
+ "step": 4780
+ },
+ {
+ "epoch": 1.2868921138184133,
+ "grad_norm": 0.2925741374492645,
+ "learning_rate": 5.3505361930294905e-06,
+ "loss": 0.5131395816802978,
+ "step": 4800
+ },
+ {
+ "epoch": 1.29225458323558,
+ "grad_norm": 0.18971531093120575,
+ "learning_rate": 5.310321715817694e-06,
+ "loss": 0.455674409866333,
+ "step": 4820
+ },
+ {
+ "epoch": 1.2976170526527466,
+ "grad_norm": 0.16778405010700226,
+ "learning_rate": 5.270107238605898e-06,
+ "loss": 0.5070962905883789,
+ "step": 4840
+ },
+ {
+ "epoch": 1.3029795220699132,
+ "grad_norm": 0.30026957392692566,
+ "learning_rate": 5.2298927613941016e-06,
+ "loss": 0.5120027542114258,
+ "step": 4860
+ },
+ {
+ "epoch": 1.3083419914870797,
+ "grad_norm": 0.17846634984016418,
+ "learning_rate": 5.189678284182305e-06,
+ "loss": 0.5114477157592774,
+ "step": 4880
+ },
+ {
+ "epoch": 1.3137044609042463,
+ "grad_norm": 0.1962418258190155,
+ "learning_rate": 5.149463806970509e-06,
+ "loss": 0.5043613910675049,
+ "step": 4900
+ },
+ {
+ "epoch": 1.319066930321413,
+ "grad_norm": 0.18446756899356842,
+ "learning_rate": 5.1092493297587135e-06,
+ "loss": 0.5396455287933349,
+ "step": 4920
+ },
+ {
+ "epoch": 1.3244293997385796,
+ "grad_norm": 0.20886844396591187,
+ "learning_rate": 5.069034852546917e-06,
+ "loss": 0.4879767417907715,
+ "step": 4940
+ },
+ {
+ "epoch": 1.3297918691557462,
+ "grad_norm": 0.16687901318073273,
+ "learning_rate": 5.028820375335121e-06,
+ "loss": 0.5014327049255372,
+ "step": 4960
+ },
+ {
+ "epoch": 1.3351543385729128,
+ "grad_norm": 0.19595153629779816,
+ "learning_rate": 4.988605898123325e-06,
+ "loss": 0.5375277996063232,
+ "step": 4980
+ },
+ {
+ "epoch": 1.3405168079900793,
+ "grad_norm": 0.2372344732284546,
+ "learning_rate": 4.948391420911528e-06,
+ "loss": 0.5020076274871826,
+ "step": 5000
+ },
+ {
+ "epoch": 1.345879277407246,
+ "grad_norm": 0.21030014753341675,
+ "learning_rate": 4.908176943699732e-06,
+ "loss": 0.5111066818237304,
+ "step": 5020
+ },
+ {
+ "epoch": 1.3512417468244127,
+ "grad_norm": 0.1866692751646042,
+ "learning_rate": 4.867962466487936e-06,
+ "loss": 0.4515383720397949,
+ "step": 5040
+ },
+ {
+ "epoch": 1.3566042162415792,
+ "grad_norm": 0.22531798481941223,
+ "learning_rate": 4.827747989276139e-06,
+ "loss": 0.4757690906524658,
+ "step": 5060
+ },
+ {
+ "epoch": 1.3619666856587458,
+ "grad_norm": 0.15868768095970154,
+ "learning_rate": 4.787533512064343e-06,
+ "loss": 0.45842318534851073,
+ "step": 5080
+ },
+ {
+ "epoch": 1.3673291550759124,
+ "grad_norm": 0.24528546631336212,
+ "learning_rate": 4.747319034852547e-06,
+ "loss": 0.47269258499145506,
+ "step": 5100
+ },
+ {
+ "epoch": 1.3726916244930791,
+ "grad_norm": 0.17387732863426208,
+ "learning_rate": 4.707104557640751e-06,
+ "loss": 0.5103805065155029,
+ "step": 5120
+ },
+ {
+ "epoch": 1.3780540939102457,
+ "grad_norm": 0.20686905086040497,
+ "learning_rate": 4.666890080428955e-06,
+ "loss": 0.5135180950164795,
+ "step": 5140
+ },
+ {
+ "epoch": 1.3834165633274123,
+ "grad_norm": 0.19599783420562744,
+ "learning_rate": 4.626675603217159e-06,
+ "loss": 0.5045839786529541,
+ "step": 5160
+ },
+ {
+ "epoch": 1.3887790327445788,
+ "grad_norm": 0.2585010528564453,
+ "learning_rate": 4.586461126005362e-06,
+ "loss": 0.45903496742248534,
+ "step": 5180
+ },
+ {
+ "epoch": 1.3941415021617454,
+ "grad_norm": 0.1688319593667984,
+ "learning_rate": 4.546246648793566e-06,
+ "loss": 0.5017509937286377,
+ "step": 5200
+ },
+ {
+ "epoch": 1.3995039715789122,
+ "grad_norm": 0.21520815789699554,
+ "learning_rate": 4.50603217158177e-06,
+ "loss": 0.48459539413452146,
+ "step": 5220
+ },
+ {
+ "epoch": 1.4048664409960787,
+ "grad_norm": 0.20514647662639618,
+ "learning_rate": 4.4658176943699735e-06,
+ "loss": 0.5073423862457276,
+ "step": 5240
+ },
+ {
+ "epoch": 1.4102289104132453,
+ "grad_norm": 0.21835413575172424,
+ "learning_rate": 4.425603217158177e-06,
+ "loss": 0.5290310382843018,
+ "step": 5260
+ },
+ {
+ "epoch": 1.4155913798304118,
+ "grad_norm": 0.28042587637901306,
+ "learning_rate": 4.385388739946381e-06,
+ "loss": 0.4823312759399414,
+ "step": 5280
+ },
+ {
+ "epoch": 1.4209538492475784,
+ "grad_norm": 0.18959026038646698,
+ "learning_rate": 4.345174262734585e-06,
+ "loss": 0.4921241760253906,
+ "step": 5300
+ },
+ {
+ "epoch": 1.4263163186647452,
+ "grad_norm": 0.18584316968917847,
+ "learning_rate": 4.304959785522788e-06,
+ "loss": 0.4892130374908447,
+ "step": 5320
+ },
+ {
+ "epoch": 1.4316787880819117,
+ "grad_norm": 0.17588038742542267,
+ "learning_rate": 4.264745308310992e-06,
+ "loss": 0.4822041988372803,
+ "step": 5340
+ },
+ {
+ "epoch": 1.4370412574990783,
+ "grad_norm": 0.18146033585071564,
+ "learning_rate": 4.224530831099196e-06,
+ "loss": 0.5084807395935058,
+ "step": 5360
+ },
+ {
+ "epoch": 1.4424037269162449,
+ "grad_norm": 0.2251797467470169,
+ "learning_rate": 4.184316353887399e-06,
+ "loss": 0.5146170139312745,
+ "step": 5380
+ },
+ {
+ "epoch": 1.4477661963334114,
+ "grad_norm": 0.18744796514511108,
+ "learning_rate": 4.144101876675603e-06,
+ "loss": 0.5189927577972412,
+ "step": 5400
+ },
+ {
+ "epoch": 1.4531286657505782,
+ "grad_norm": 0.25737133622169495,
+ "learning_rate": 4.103887399463807e-06,
+ "loss": 0.4891658782958984,
+ "step": 5420
+ },
+ {
+ "epoch": 1.4584911351677448,
+ "grad_norm": 0.20580479502677917,
+ "learning_rate": 4.0636729222520105e-06,
+ "loss": 0.4953591823577881,
+ "step": 5440
+ },
+ {
+ "epoch": 1.4638536045849113,
+ "grad_norm": 0.2351546287536621,
+ "learning_rate": 4.023458445040214e-06,
+ "loss": 0.5025320053100586,
+ "step": 5460
+ },
+ {
+ "epoch": 1.4692160740020779,
+ "grad_norm": 0.1819481998682022,
+ "learning_rate": 3.983243967828418e-06,
+ "loss": 0.47151756286621094,
+ "step": 5480
+ },
+ {
+ "epoch": 1.4745785434192444,
+ "grad_norm": 0.20772472023963928,
+ "learning_rate": 3.943029490616622e-06,
+ "loss": 0.4678915023803711,
+ "step": 5500
+ },
+ {
+ "epoch": 1.4799410128364112,
+ "grad_norm": 0.2203037440776825,
+ "learning_rate": 3.902815013404825e-06,
+ "loss": 0.46007452011108396,
+ "step": 5520
+ },
+ {
+ "epoch": 1.4853034822535778,
+ "grad_norm": 0.15371400117874146,
+ "learning_rate": 3.86260053619303e-06,
+ "loss": 0.44407024383544924,
+ "step": 5540
+ },
+ {
+ "epoch": 1.4906659516707443,
+ "grad_norm": 0.2276080846786499,
+ "learning_rate": 3.8223860589812335e-06,
+ "loss": 0.4730556488037109,
+ "step": 5560
+ },
+ {
+ "epoch": 1.4960284210879111,
+ "grad_norm": 0.24482466280460358,
+ "learning_rate": 3.7821715817694376e-06,
+ "loss": 0.5073911666870117,
+ "step": 5580
+ },
+ {
+ "epoch": 1.5013908905050775,
+ "grad_norm": 0.20438458025455475,
+ "learning_rate": 3.741957104557641e-06,
+ "loss": 0.46701641082763673,
+ "step": 5600
+ },
+ {
+ "epoch": 1.5067533599222442,
+ "grad_norm": 0.19854313135147095,
+ "learning_rate": 3.7017426273458446e-06,
+ "loss": 0.46309399604797363,
+ "step": 5620
+ },
+ {
+ "epoch": 1.5121158293394108,
+ "grad_norm": 0.18356069922447205,
+ "learning_rate": 3.6615281501340483e-06,
+ "loss": 0.503613805770874,
+ "step": 5640
+ },
+ {
+ "epoch": 1.5174782987565774,
+ "grad_norm": 0.2009744495153427,
+ "learning_rate": 3.621313672922252e-06,
+ "loss": 0.4765054225921631,
+ "step": 5660
+ },
+ {
+ "epoch": 1.5228407681737441,
+ "grad_norm": 0.3058745563030243,
+ "learning_rate": 3.5810991957104557e-06,
+ "loss": 0.5179148197174073,
+ "step": 5680
+ },
+ {
+ "epoch": 1.5282032375909105,
+ "grad_norm": 0.17671597003936768,
+ "learning_rate": 3.54088471849866e-06,
+ "loss": 0.45907344818115237,
+ "step": 5700
+ },
+ {
+ "epoch": 1.5335657070080773,
+ "grad_norm": 0.22209160029888153,
+ "learning_rate": 3.5006702412868635e-06,
+ "loss": 0.49304862022399903,
+ "step": 5720
+ },
+ {
+ "epoch": 1.5389281764252438,
+ "grad_norm": 0.21018914878368378,
+ "learning_rate": 3.4604557640750672e-06,
+ "loss": 0.5536758422851562,
+ "step": 5740
+ },
+ {
+ "epoch": 1.5442906458424104,
+ "grad_norm": 0.14339996874332428,
+ "learning_rate": 3.420241286863271e-06,
+ "loss": 0.48726091384887693,
+ "step": 5760
+ },
+ {
+ "epoch": 1.5496531152595772,
+ "grad_norm": 0.11419746279716492,
+ "learning_rate": 3.3800268096514746e-06,
+ "loss": 0.4514151573181152,
+ "step": 5780
+ },
+ {
+ "epoch": 1.5550155846767435,
+ "grad_norm": 0.18168962001800537,
+ "learning_rate": 3.3398123324396783e-06,
+ "loss": 0.5279990196228027,
+ "step": 5800
+ },
+ {
+ "epoch": 1.5603780540939103,
+ "grad_norm": 0.24244488775730133,
+ "learning_rate": 3.299597855227882e-06,
+ "loss": 0.49297361373901366,
+ "step": 5820
+ },
+ {
+ "epoch": 1.5657405235110768,
+ "grad_norm": 0.2017296999692917,
+ "learning_rate": 3.2593833780160857e-06,
+ "loss": 0.49305019378662107,
+ "step": 5840
+ },
+ {
+ "epoch": 1.5711029929282434,
+ "grad_norm": 0.22592377662658691,
+ "learning_rate": 3.2191689008042894e-06,
+ "loss": 0.4862989902496338,
+ "step": 5860
+ },
+ {
+ "epoch": 1.5764654623454102,
+ "grad_norm": 0.24772357940673828,
+ "learning_rate": 3.1789544235924935e-06,
+ "loss": 0.45182647705078127,
+ "step": 5880
+ },
+ {
+ "epoch": 1.5818279317625765,
+ "grad_norm": 0.20607218146324158,
+ "learning_rate": 3.1387399463806972e-06,
+ "loss": 0.48905248641967775,
+ "step": 5900
+ },
+ {
+ "epoch": 1.5871904011797433,
+ "grad_norm": 0.1931353509426117,
+ "learning_rate": 3.098525469168901e-06,
+ "loss": 0.5307461261749268,
+ "step": 5920
+ },
+ {
+ "epoch": 1.5925528705969099,
+ "grad_norm": 0.16020581126213074,
+ "learning_rate": 3.0583109919571046e-06,
+ "loss": 0.4672811985015869,
+ "step": 5940
+ },
+ {
+ "epoch": 1.5979153400140764,
+ "grad_norm": 0.23668015003204346,
+ "learning_rate": 3.0180965147453083e-06,
+ "loss": 0.5272688865661621,
+ "step": 5960
+ },
+ {
+ "epoch": 1.6032778094312432,
+ "grad_norm": 0.1916576772928238,
+ "learning_rate": 2.977882037533512e-06,
+ "loss": 0.4859332084655762,
+ "step": 5980
+ },
+ {
+ "epoch": 1.6086402788484095,
+ "grad_norm": 0.23635101318359375,
+ "learning_rate": 2.9376675603217157e-06,
+ "loss": 0.5418910980224609,
+ "step": 6000
+ },
+ {
+ "epoch": 1.6140027482655763,
+ "grad_norm": 0.2404562532901764,
+ "learning_rate": 2.89745308310992e-06,
+ "loss": 0.5449445247650146,
+ "step": 6020
+ },
+ {
+ "epoch": 1.6193652176827429,
+ "grad_norm": 0.20147347450256348,
+ "learning_rate": 2.8572386058981235e-06,
+ "loss": 0.4737790584564209,
+ "step": 6040
+ },
+ {
+ "epoch": 1.6247276870999094,
+ "grad_norm": 0.2455863654613495,
+ "learning_rate": 2.8170241286863272e-06,
+ "loss": 0.4722298145294189,
+ "step": 6060
+ },
+ {
+ "epoch": 1.6300901565170762,
+ "grad_norm": 0.22172148525714874,
+ "learning_rate": 2.776809651474531e-06,
+ "loss": 0.5120372295379638,
+ "step": 6080
+ },
+ {
+ "epoch": 1.6354526259342426,
+ "grad_norm": 0.3848462700843811,
+ "learning_rate": 2.7365951742627346e-06,
+ "loss": 0.5152206897735596,
+ "step": 6100
+ },
+ {
+ "epoch": 1.6408150953514093,
+ "grad_norm": 0.19071047008037567,
+ "learning_rate": 2.6963806970509383e-06,
+ "loss": 0.4757692813873291,
+ "step": 6120
+ },
+ {
+ "epoch": 1.646177564768576,
+ "grad_norm": 0.20568661391735077,
+ "learning_rate": 2.656166219839142e-06,
+ "loss": 0.475917387008667,
+ "step": 6140
+ },
+ {
+ "epoch": 1.6515400341857425,
+ "grad_norm": 0.11777322739362717,
+ "learning_rate": 2.6159517426273457e-06,
+ "loss": 0.5161296367645264,
+ "step": 6160
+ },
+ {
+ "epoch": 1.6569025036029092,
+ "grad_norm": 0.1700555831193924,
+ "learning_rate": 2.5757372654155494e-06,
+ "loss": 0.4715432167053223,
+ "step": 6180
+ },
+ {
+ "epoch": 1.6622649730200756,
+ "grad_norm": 0.18927083909511566,
+ "learning_rate": 2.5355227882037535e-06,
+ "loss": 0.49937710762023924,
+ "step": 6200
+ },
+ {
+ "epoch": 1.6676274424372424,
+ "grad_norm": 0.22097784280776978,
+ "learning_rate": 2.4953083109919572e-06,
+ "loss": 0.43366107940673826,
+ "step": 6220
+ },
+ {
+ "epoch": 1.672989911854409,
+ "grad_norm": 0.2299281805753708,
+ "learning_rate": 2.455093833780161e-06,
+ "loss": 0.5145821094512939,
+ "step": 6240
+ },
+ {
+ "epoch": 1.6783523812715755,
+ "grad_norm": 0.2384844720363617,
+ "learning_rate": 2.4148793565683646e-06,
+ "loss": 0.459308385848999,
+ "step": 6260
+ },
+ {
+ "epoch": 1.6837148506887423,
+ "grad_norm": 0.24471035599708557,
+ "learning_rate": 2.3746648793565683e-06,
+ "loss": 0.4676504611968994,
+ "step": 6280
+ },
+ {
+ "epoch": 1.6890773201059086,
+ "grad_norm": 0.24419866502285004,
+ "learning_rate": 2.334450402144772e-06,
+ "loss": 0.4745138168334961,
+ "step": 6300
+ },
+ {
+ "epoch": 1.6944397895230754,
+ "grad_norm": 0.15896575152873993,
+ "learning_rate": 2.294235924932976e-06,
+ "loss": 0.5073649883270264,
+ "step": 6320
+ },
+ {
+ "epoch": 1.699802258940242,
+ "grad_norm": 0.26504868268966675,
+ "learning_rate": 2.25402144772118e-06,
+ "loss": 0.4534353733062744,
+ "step": 6340
+ },
+ {
+ "epoch": 1.7051647283574085,
+ "grad_norm": 0.2461850792169571,
+ "learning_rate": 2.2138069705093836e-06,
+ "loss": 0.4862947940826416,
+ "step": 6360
+ },
+ {
+ "epoch": 1.7105271977745753,
+ "grad_norm": 0.17332817614078522,
+ "learning_rate": 2.1735924932975873e-06,
+ "loss": 0.5049370765686035,
+ "step": 6380
+ },
+ {
+ "epoch": 1.7158896671917419,
+ "grad_norm": 0.19762548804283142,
+ "learning_rate": 2.133378016085791e-06,
+ "loss": 0.5272616386413574,
+ "step": 6400
+ },
+ {
+ "epoch": 1.7212521366089084,
+ "grad_norm": 0.23265399038791656,
+ "learning_rate": 2.0931635388739946e-06,
+ "loss": 0.47600841522216797,
+ "step": 6420
+ },
+ {
+ "epoch": 1.726614606026075,
+ "grad_norm": 0.20868578553199768,
+ "learning_rate": 2.0529490616621983e-06,
+ "loss": 0.5027226448059082,
+ "step": 6440
+ },
+ {
+ "epoch": 1.7319770754432415,
+ "grad_norm": 0.2851981520652771,
+ "learning_rate": 2.012734584450402e-06,
+ "loss": 0.5288124561309815,
+ "step": 6460
+ },
+ {
+ "epoch": 1.7373395448604083,
+ "grad_norm": 0.20086587965488434,
+ "learning_rate": 1.9725201072386057e-06,
+ "loss": 0.4625516891479492,
+ "step": 6480
+ },
+ {
+ "epoch": 1.7427020142775749,
+ "grad_norm": 0.24060192704200745,
+ "learning_rate": 1.93230563002681e-06,
+ "loss": 0.4843903541564941,
+ "step": 6500
+ },
+ {
+ "epoch": 1.7480644836947414,
+ "grad_norm": 0.33561915159225464,
+ "learning_rate": 1.8920911528150133e-06,
+ "loss": 0.4823720932006836,
+ "step": 6520
+ },
+ {
+ "epoch": 1.753426953111908,
+ "grad_norm": 0.2510465383529663,
+ "learning_rate": 1.851876675603217e-06,
+ "loss": 0.46517143249511717,
+ "step": 6540
+ },
+ {
+ "epoch": 1.7587894225290746,
+ "grad_norm": 0.2631177604198456,
+ "learning_rate": 1.811662198391421e-06,
+ "loss": 0.5004732131958007,
+ "step": 6560
+ },
+ {
+ "epoch": 1.7641518919462413,
+ "grad_norm": 0.3493230640888214,
+ "learning_rate": 1.7714477211796249e-06,
+ "loss": 0.523811674118042,
+ "step": 6580
+ },
+ {
+ "epoch": 1.769514361363408,
+ "grad_norm": 0.1742691546678543,
+ "learning_rate": 1.7312332439678286e-06,
+ "loss": 0.5276295661926269,
+ "step": 6600
+ },
+ {
+ "epoch": 1.7748768307805745,
+ "grad_norm": 0.16134823858737946,
+ "learning_rate": 1.6910187667560323e-06,
+ "loss": 0.5352637290954589,
+ "step": 6620
+ },
+ {
+ "epoch": 1.780239300197741,
+ "grad_norm": 0.20977018773555756,
+ "learning_rate": 1.650804289544236e-06,
+ "loss": 0.4955774784088135,
+ "step": 6640
+ },
+ {
+ "epoch": 1.7856017696149076,
+ "grad_norm": 0.20511005818843842,
+ "learning_rate": 1.6105898123324397e-06,
+ "loss": 0.48643174171447756,
+ "step": 6660
+ },
+ {
+ "epoch": 1.7909642390320744,
+ "grad_norm": 0.23870044946670532,
+ "learning_rate": 1.5703753351206434e-06,
+ "loss": 0.4673162460327148,
+ "step": 6680
+ },
+ {
+ "epoch": 1.796326708449241,
+ "grad_norm": 0.21660065650939941,
+ "learning_rate": 1.5301608579088473e-06,
+ "loss": 0.5381903648376465,
+ "step": 6700
+ },
+ {
+ "epoch": 1.8016891778664075,
+ "grad_norm": 0.26977139711380005,
+ "learning_rate": 1.489946380697051e-06,
+ "loss": 0.42094998359680175,
+ "step": 6720
+ },
+ {
+ "epoch": 1.807051647283574,
+ "grad_norm": 0.2088550478219986,
+ "learning_rate": 1.4497319034852549e-06,
+ "loss": 0.49211792945861815,
+ "step": 6740
+ },
+ {
+ "epoch": 1.8124141167007406,
+ "grad_norm": 0.18141885101795197,
+ "learning_rate": 1.4095174262734586e-06,
+ "loss": 0.46572179794311525,
+ "step": 6760
+ },
+ {
+ "epoch": 1.8177765861179074,
+ "grad_norm": 0.2200685739517212,
+ "learning_rate": 1.3693029490616623e-06,
+ "loss": 0.4996177196502686,
+ "step": 6780
+ },
+ {
+ "epoch": 1.823139055535074,
+ "grad_norm": 0.19545452296733856,
+ "learning_rate": 1.329088471849866e-06,
+ "loss": 0.4731945514678955,
+ "step": 6800
+ },
+ {
+ "epoch": 1.8285015249522405,
+ "grad_norm": 0.2239731252193451,
+ "learning_rate": 1.2888739946380697e-06,
+ "loss": 0.47544050216674805,
+ "step": 6820
+ },
+ {
+ "epoch": 1.833863994369407,
+ "grad_norm": 0.22336581349372864,
+ "learning_rate": 1.2486595174262734e-06,
+ "loss": 0.47878737449645997,
+ "step": 6840
+ },
+ {
+ "epoch": 1.8392264637865736,
+ "grad_norm": 0.20921571552753448,
+ "learning_rate": 1.2084450402144773e-06,
+ "loss": 0.41347403526306153,
+ "step": 6860
+ },
+ {
+ "epoch": 1.8445889332037404,
+ "grad_norm": 0.1577194333076477,
+ "learning_rate": 1.168230563002681e-06,
+ "loss": 0.5464958667755127,
+ "step": 6880
+ },
+ {
+ "epoch": 1.849951402620907,
+ "grad_norm": 0.1477355808019638,
+ "learning_rate": 1.1280160857908849e-06,
+ "loss": 0.48548617362976076,
+ "step": 6900
+ },
+ {
+ "epoch": 1.8553138720380735,
+ "grad_norm": 0.22352682054042816,
+ "learning_rate": 1.0878016085790886e-06,
+ "loss": 0.4518588542938232,
+ "step": 6920
+ },
+ {
+ "epoch": 1.8606763414552403,
+ "grad_norm": 0.19822706282138824,
+ "learning_rate": 1.0475871313672923e-06,
+ "loss": 0.4190972805023193,
+ "step": 6940
+ },
+ {
+ "epoch": 1.8660388108724066,
+ "grad_norm": 0.20670010149478912,
+ "learning_rate": 1.007372654155496e-06,
+ "loss": 0.5045090675354004,
+ "step": 6960
+ },
+ {
+ "epoch": 1.8714012802895734,
+ "grad_norm": 0.2154514342546463,
+ "learning_rate": 9.671581769436997e-07,
+ "loss": 0.4472982883453369,
+ "step": 6980
+ },
+ {
+ "epoch": 1.87676374970674,
+ "grad_norm": 0.19451302289962769,
+ "learning_rate": 9.269436997319035e-07,
+ "loss": 0.4328409194946289,
+ "step": 7000
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 8.620995010015519e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-7000/training_args.bin b/checkpoint-7000/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-7000/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-7200/README.md b/checkpoint-7200/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-7200/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-7200/adapter_config.json b/checkpoint-7200/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-7200/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-7200/adapter_model.safetensors b/checkpoint-7200/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..9f055fc8abdaec3c7b782514e3c1542994d68cff
--- /dev/null
+++ b/checkpoint-7200/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5008c50618f5d0ed327d88a93d41f3cdc02a97240319db261643e05167eb465a
+size 50360752
diff --git a/checkpoint-7200/chat_template.jinja b/checkpoint-7200/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-7200/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-7200/optimizer.pt b/checkpoint-7200/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..38adf2b7caee9e427f28df019f90fe63b53e8a87
--- /dev/null
+++ b/checkpoint-7200/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a8585e77057ed91d1a6749c6045fe9717378b70a3650a76c0fecbc92a19f3735
+size 100828235
diff --git a/checkpoint-7200/rng_state.pth b/checkpoint-7200/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..6dbe3dd1a22eb0152d28f4f426a96fb325e714c2
--- /dev/null
+++ b/checkpoint-7200/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:993ebcc8811681c7cc30a6b9b00d1bad83cd6154bfbb94372a71b16f140254bf
+size 14645
diff --git a/checkpoint-7200/scheduler.pt b/checkpoint-7200/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..4a2886cad7602acc88568e1df58d4ec35ec2fb85
--- /dev/null
+++ b/checkpoint-7200/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:bdfe5d3784208321e032d0e5bb471a989eb069b864e2db02885ffb5be3116a81
+size 1465
diff --git a/checkpoint-7200/tokenizer.json b/checkpoint-7200/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-7200/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-7200/tokenizer_config.json b/checkpoint-7200/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-7200/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-7200/trainer_state.json b/checkpoint-7200/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..869ea6a497cdbcaa6d64c2e331f311b08a746887
--- /dev/null
+++ b/checkpoint-7200/trainer_state.json
@@ -0,0 +1,2554 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.930388443878406,
+ "eval_steps": 500,
+ "global_step": 7200,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ },
+ {
+ "epoch": 1.185005194892248,
+ "grad_norm": 0.16667844355106354,
+ "learning_rate": 6.114611260053619e-06,
+ "loss": 0.5159809589385986,
+ "step": 4420
+ },
+ {
+ "epoch": 1.1903676643094145,
+ "grad_norm": 0.19206570088863373,
+ "learning_rate": 6.074396782841823e-06,
+ "loss": 0.47541089057922364,
+ "step": 4440
+ },
+ {
+ "epoch": 1.195730133726581,
+ "grad_norm": 0.17394617199897766,
+ "learning_rate": 6.034182305630027e-06,
+ "loss": 0.5470661640167236,
+ "step": 4460
+ },
+ {
+ "epoch": 1.2010926031437477,
+ "grad_norm": 0.210404634475708,
+ "learning_rate": 5.993967828418231e-06,
+ "loss": 0.5377882957458496,
+ "step": 4480
+ },
+ {
+ "epoch": 1.2064550725609142,
+ "grad_norm": 0.18084648251533508,
+ "learning_rate": 5.953753351206435e-06,
+ "loss": 0.5037185192108155,
+ "step": 4500
+ },
+ {
+ "epoch": 1.211817541978081,
+ "grad_norm": 0.23707027733325958,
+ "learning_rate": 5.913538873994639e-06,
+ "loss": 0.4822190284729004,
+ "step": 4520
+ },
+ {
+ "epoch": 1.2171800113952476,
+ "grad_norm": 0.16474473476409912,
+ "learning_rate": 5.873324396782842e-06,
+ "loss": 0.46645288467407225,
+ "step": 4540
+ },
+ {
+ "epoch": 1.2225424808124141,
+ "grad_norm": 0.2142348438501358,
+ "learning_rate": 5.833109919571046e-06,
+ "loss": 0.5255855560302735,
+ "step": 4560
+ },
+ {
+ "epoch": 1.2279049502295807,
+ "grad_norm": 0.2531765103340149,
+ "learning_rate": 5.79289544235925e-06,
+ "loss": 0.507044792175293,
+ "step": 4580
+ },
+ {
+ "epoch": 1.2332674196467472,
+ "grad_norm": 0.2553550899028778,
+ "learning_rate": 5.7526809651474535e-06,
+ "loss": 0.4767824649810791,
+ "step": 4600
+ },
+ {
+ "epoch": 1.238629889063914,
+ "grad_norm": 0.14484412968158722,
+ "learning_rate": 5.712466487935657e-06,
+ "loss": 0.4675601005554199,
+ "step": 4620
+ },
+ {
+ "epoch": 1.2439923584810806,
+ "grad_norm": 0.14328251779079437,
+ "learning_rate": 5.672252010723861e-06,
+ "loss": 0.4956005573272705,
+ "step": 4640
+ },
+ {
+ "epoch": 1.2493548278982471,
+ "grad_norm": 0.1739245355129242,
+ "learning_rate": 5.632037533512065e-06,
+ "loss": 0.48583345413208007,
+ "step": 4660
+ },
+ {
+ "epoch": 1.2547172973154137,
+ "grad_norm": 0.21294184029102325,
+ "learning_rate": 5.591823056300268e-06,
+ "loss": 0.520921277999878,
+ "step": 4680
+ },
+ {
+ "epoch": 1.2600797667325803,
+ "grad_norm": 0.25132355093955994,
+ "learning_rate": 5.551608579088472e-06,
+ "loss": 0.5295385837554931,
+ "step": 4700
+ },
+ {
+ "epoch": 1.265442236149747,
+ "grad_norm": 0.18603841960430145,
+ "learning_rate": 5.511394101876676e-06,
+ "loss": 0.47570199966430665,
+ "step": 4720
+ },
+ {
+ "epoch": 1.2708047055669136,
+ "grad_norm": 0.19883134961128235,
+ "learning_rate": 5.471179624664879e-06,
+ "loss": 0.5016080379486084,
+ "step": 4740
+ },
+ {
+ "epoch": 1.2761671749840802,
+ "grad_norm": 0.19640181958675385,
+ "learning_rate": 5.430965147453083e-06,
+ "loss": 0.4999081134796143,
+ "step": 4760
+ },
+ {
+ "epoch": 1.2815296444012467,
+ "grad_norm": 0.2584764361381531,
+ "learning_rate": 5.390750670241287e-06,
+ "loss": 0.4780082702636719,
+ "step": 4780
+ },
+ {
+ "epoch": 1.2868921138184133,
+ "grad_norm": 0.2925741374492645,
+ "learning_rate": 5.3505361930294905e-06,
+ "loss": 0.5131395816802978,
+ "step": 4800
+ },
+ {
+ "epoch": 1.29225458323558,
+ "grad_norm": 0.18971531093120575,
+ "learning_rate": 5.310321715817694e-06,
+ "loss": 0.455674409866333,
+ "step": 4820
+ },
+ {
+ "epoch": 1.2976170526527466,
+ "grad_norm": 0.16778405010700226,
+ "learning_rate": 5.270107238605898e-06,
+ "loss": 0.5070962905883789,
+ "step": 4840
+ },
+ {
+ "epoch": 1.3029795220699132,
+ "grad_norm": 0.30026957392692566,
+ "learning_rate": 5.2298927613941016e-06,
+ "loss": 0.5120027542114258,
+ "step": 4860
+ },
+ {
+ "epoch": 1.3083419914870797,
+ "grad_norm": 0.17846634984016418,
+ "learning_rate": 5.189678284182305e-06,
+ "loss": 0.5114477157592774,
+ "step": 4880
+ },
+ {
+ "epoch": 1.3137044609042463,
+ "grad_norm": 0.1962418258190155,
+ "learning_rate": 5.149463806970509e-06,
+ "loss": 0.5043613910675049,
+ "step": 4900
+ },
+ {
+ "epoch": 1.319066930321413,
+ "grad_norm": 0.18446756899356842,
+ "learning_rate": 5.1092493297587135e-06,
+ "loss": 0.5396455287933349,
+ "step": 4920
+ },
+ {
+ "epoch": 1.3244293997385796,
+ "grad_norm": 0.20886844396591187,
+ "learning_rate": 5.069034852546917e-06,
+ "loss": 0.4879767417907715,
+ "step": 4940
+ },
+ {
+ "epoch": 1.3297918691557462,
+ "grad_norm": 0.16687901318073273,
+ "learning_rate": 5.028820375335121e-06,
+ "loss": 0.5014327049255372,
+ "step": 4960
+ },
+ {
+ "epoch": 1.3351543385729128,
+ "grad_norm": 0.19595153629779816,
+ "learning_rate": 4.988605898123325e-06,
+ "loss": 0.5375277996063232,
+ "step": 4980
+ },
+ {
+ "epoch": 1.3405168079900793,
+ "grad_norm": 0.2372344732284546,
+ "learning_rate": 4.948391420911528e-06,
+ "loss": 0.5020076274871826,
+ "step": 5000
+ },
+ {
+ "epoch": 1.345879277407246,
+ "grad_norm": 0.21030014753341675,
+ "learning_rate": 4.908176943699732e-06,
+ "loss": 0.5111066818237304,
+ "step": 5020
+ },
+ {
+ "epoch": 1.3512417468244127,
+ "grad_norm": 0.1866692751646042,
+ "learning_rate": 4.867962466487936e-06,
+ "loss": 0.4515383720397949,
+ "step": 5040
+ },
+ {
+ "epoch": 1.3566042162415792,
+ "grad_norm": 0.22531798481941223,
+ "learning_rate": 4.827747989276139e-06,
+ "loss": 0.4757690906524658,
+ "step": 5060
+ },
+ {
+ "epoch": 1.3619666856587458,
+ "grad_norm": 0.15868768095970154,
+ "learning_rate": 4.787533512064343e-06,
+ "loss": 0.45842318534851073,
+ "step": 5080
+ },
+ {
+ "epoch": 1.3673291550759124,
+ "grad_norm": 0.24528546631336212,
+ "learning_rate": 4.747319034852547e-06,
+ "loss": 0.47269258499145506,
+ "step": 5100
+ },
+ {
+ "epoch": 1.3726916244930791,
+ "grad_norm": 0.17387732863426208,
+ "learning_rate": 4.707104557640751e-06,
+ "loss": 0.5103805065155029,
+ "step": 5120
+ },
+ {
+ "epoch": 1.3780540939102457,
+ "grad_norm": 0.20686905086040497,
+ "learning_rate": 4.666890080428955e-06,
+ "loss": 0.5135180950164795,
+ "step": 5140
+ },
+ {
+ "epoch": 1.3834165633274123,
+ "grad_norm": 0.19599783420562744,
+ "learning_rate": 4.626675603217159e-06,
+ "loss": 0.5045839786529541,
+ "step": 5160
+ },
+ {
+ "epoch": 1.3887790327445788,
+ "grad_norm": 0.2585010528564453,
+ "learning_rate": 4.586461126005362e-06,
+ "loss": 0.45903496742248534,
+ "step": 5180
+ },
+ {
+ "epoch": 1.3941415021617454,
+ "grad_norm": 0.1688319593667984,
+ "learning_rate": 4.546246648793566e-06,
+ "loss": 0.5017509937286377,
+ "step": 5200
+ },
+ {
+ "epoch": 1.3995039715789122,
+ "grad_norm": 0.21520815789699554,
+ "learning_rate": 4.50603217158177e-06,
+ "loss": 0.48459539413452146,
+ "step": 5220
+ },
+ {
+ "epoch": 1.4048664409960787,
+ "grad_norm": 0.20514647662639618,
+ "learning_rate": 4.4658176943699735e-06,
+ "loss": 0.5073423862457276,
+ "step": 5240
+ },
+ {
+ "epoch": 1.4102289104132453,
+ "grad_norm": 0.21835413575172424,
+ "learning_rate": 4.425603217158177e-06,
+ "loss": 0.5290310382843018,
+ "step": 5260
+ },
+ {
+ "epoch": 1.4155913798304118,
+ "grad_norm": 0.28042587637901306,
+ "learning_rate": 4.385388739946381e-06,
+ "loss": 0.4823312759399414,
+ "step": 5280
+ },
+ {
+ "epoch": 1.4209538492475784,
+ "grad_norm": 0.18959026038646698,
+ "learning_rate": 4.345174262734585e-06,
+ "loss": 0.4921241760253906,
+ "step": 5300
+ },
+ {
+ "epoch": 1.4263163186647452,
+ "grad_norm": 0.18584316968917847,
+ "learning_rate": 4.304959785522788e-06,
+ "loss": 0.4892130374908447,
+ "step": 5320
+ },
+ {
+ "epoch": 1.4316787880819117,
+ "grad_norm": 0.17588038742542267,
+ "learning_rate": 4.264745308310992e-06,
+ "loss": 0.4822041988372803,
+ "step": 5340
+ },
+ {
+ "epoch": 1.4370412574990783,
+ "grad_norm": 0.18146033585071564,
+ "learning_rate": 4.224530831099196e-06,
+ "loss": 0.5084807395935058,
+ "step": 5360
+ },
+ {
+ "epoch": 1.4424037269162449,
+ "grad_norm": 0.2251797467470169,
+ "learning_rate": 4.184316353887399e-06,
+ "loss": 0.5146170139312745,
+ "step": 5380
+ },
+ {
+ "epoch": 1.4477661963334114,
+ "grad_norm": 0.18744796514511108,
+ "learning_rate": 4.144101876675603e-06,
+ "loss": 0.5189927577972412,
+ "step": 5400
+ },
+ {
+ "epoch": 1.4531286657505782,
+ "grad_norm": 0.25737133622169495,
+ "learning_rate": 4.103887399463807e-06,
+ "loss": 0.4891658782958984,
+ "step": 5420
+ },
+ {
+ "epoch": 1.4584911351677448,
+ "grad_norm": 0.20580479502677917,
+ "learning_rate": 4.0636729222520105e-06,
+ "loss": 0.4953591823577881,
+ "step": 5440
+ },
+ {
+ "epoch": 1.4638536045849113,
+ "grad_norm": 0.2351546287536621,
+ "learning_rate": 4.023458445040214e-06,
+ "loss": 0.5025320053100586,
+ "step": 5460
+ },
+ {
+ "epoch": 1.4692160740020779,
+ "grad_norm": 0.1819481998682022,
+ "learning_rate": 3.983243967828418e-06,
+ "loss": 0.47151756286621094,
+ "step": 5480
+ },
+ {
+ "epoch": 1.4745785434192444,
+ "grad_norm": 0.20772472023963928,
+ "learning_rate": 3.943029490616622e-06,
+ "loss": 0.4678915023803711,
+ "step": 5500
+ },
+ {
+ "epoch": 1.4799410128364112,
+ "grad_norm": 0.2203037440776825,
+ "learning_rate": 3.902815013404825e-06,
+ "loss": 0.46007452011108396,
+ "step": 5520
+ },
+ {
+ "epoch": 1.4853034822535778,
+ "grad_norm": 0.15371400117874146,
+ "learning_rate": 3.86260053619303e-06,
+ "loss": 0.44407024383544924,
+ "step": 5540
+ },
+ {
+ "epoch": 1.4906659516707443,
+ "grad_norm": 0.2276080846786499,
+ "learning_rate": 3.8223860589812335e-06,
+ "loss": 0.4730556488037109,
+ "step": 5560
+ },
+ {
+ "epoch": 1.4960284210879111,
+ "grad_norm": 0.24482466280460358,
+ "learning_rate": 3.7821715817694376e-06,
+ "loss": 0.5073911666870117,
+ "step": 5580
+ },
+ {
+ "epoch": 1.5013908905050775,
+ "grad_norm": 0.20438458025455475,
+ "learning_rate": 3.741957104557641e-06,
+ "loss": 0.46701641082763673,
+ "step": 5600
+ },
+ {
+ "epoch": 1.5067533599222442,
+ "grad_norm": 0.19854313135147095,
+ "learning_rate": 3.7017426273458446e-06,
+ "loss": 0.46309399604797363,
+ "step": 5620
+ },
+ {
+ "epoch": 1.5121158293394108,
+ "grad_norm": 0.18356069922447205,
+ "learning_rate": 3.6615281501340483e-06,
+ "loss": 0.503613805770874,
+ "step": 5640
+ },
+ {
+ "epoch": 1.5174782987565774,
+ "grad_norm": 0.2009744495153427,
+ "learning_rate": 3.621313672922252e-06,
+ "loss": 0.4765054225921631,
+ "step": 5660
+ },
+ {
+ "epoch": 1.5228407681737441,
+ "grad_norm": 0.3058745563030243,
+ "learning_rate": 3.5810991957104557e-06,
+ "loss": 0.5179148197174073,
+ "step": 5680
+ },
+ {
+ "epoch": 1.5282032375909105,
+ "grad_norm": 0.17671597003936768,
+ "learning_rate": 3.54088471849866e-06,
+ "loss": 0.45907344818115237,
+ "step": 5700
+ },
+ {
+ "epoch": 1.5335657070080773,
+ "grad_norm": 0.22209160029888153,
+ "learning_rate": 3.5006702412868635e-06,
+ "loss": 0.49304862022399903,
+ "step": 5720
+ },
+ {
+ "epoch": 1.5389281764252438,
+ "grad_norm": 0.21018914878368378,
+ "learning_rate": 3.4604557640750672e-06,
+ "loss": 0.5536758422851562,
+ "step": 5740
+ },
+ {
+ "epoch": 1.5442906458424104,
+ "grad_norm": 0.14339996874332428,
+ "learning_rate": 3.420241286863271e-06,
+ "loss": 0.48726091384887693,
+ "step": 5760
+ },
+ {
+ "epoch": 1.5496531152595772,
+ "grad_norm": 0.11419746279716492,
+ "learning_rate": 3.3800268096514746e-06,
+ "loss": 0.4514151573181152,
+ "step": 5780
+ },
+ {
+ "epoch": 1.5550155846767435,
+ "grad_norm": 0.18168962001800537,
+ "learning_rate": 3.3398123324396783e-06,
+ "loss": 0.5279990196228027,
+ "step": 5800
+ },
+ {
+ "epoch": 1.5603780540939103,
+ "grad_norm": 0.24244488775730133,
+ "learning_rate": 3.299597855227882e-06,
+ "loss": 0.49297361373901366,
+ "step": 5820
+ },
+ {
+ "epoch": 1.5657405235110768,
+ "grad_norm": 0.2017296999692917,
+ "learning_rate": 3.2593833780160857e-06,
+ "loss": 0.49305019378662107,
+ "step": 5840
+ },
+ {
+ "epoch": 1.5711029929282434,
+ "grad_norm": 0.22592377662658691,
+ "learning_rate": 3.2191689008042894e-06,
+ "loss": 0.4862989902496338,
+ "step": 5860
+ },
+ {
+ "epoch": 1.5764654623454102,
+ "grad_norm": 0.24772357940673828,
+ "learning_rate": 3.1789544235924935e-06,
+ "loss": 0.45182647705078127,
+ "step": 5880
+ },
+ {
+ "epoch": 1.5818279317625765,
+ "grad_norm": 0.20607218146324158,
+ "learning_rate": 3.1387399463806972e-06,
+ "loss": 0.48905248641967775,
+ "step": 5900
+ },
+ {
+ "epoch": 1.5871904011797433,
+ "grad_norm": 0.1931353509426117,
+ "learning_rate": 3.098525469168901e-06,
+ "loss": 0.5307461261749268,
+ "step": 5920
+ },
+ {
+ "epoch": 1.5925528705969099,
+ "grad_norm": 0.16020581126213074,
+ "learning_rate": 3.0583109919571046e-06,
+ "loss": 0.4672811985015869,
+ "step": 5940
+ },
+ {
+ "epoch": 1.5979153400140764,
+ "grad_norm": 0.23668015003204346,
+ "learning_rate": 3.0180965147453083e-06,
+ "loss": 0.5272688865661621,
+ "step": 5960
+ },
+ {
+ "epoch": 1.6032778094312432,
+ "grad_norm": 0.1916576772928238,
+ "learning_rate": 2.977882037533512e-06,
+ "loss": 0.4859332084655762,
+ "step": 5980
+ },
+ {
+ "epoch": 1.6086402788484095,
+ "grad_norm": 0.23635101318359375,
+ "learning_rate": 2.9376675603217157e-06,
+ "loss": 0.5418910980224609,
+ "step": 6000
+ },
+ {
+ "epoch": 1.6140027482655763,
+ "grad_norm": 0.2404562532901764,
+ "learning_rate": 2.89745308310992e-06,
+ "loss": 0.5449445247650146,
+ "step": 6020
+ },
+ {
+ "epoch": 1.6193652176827429,
+ "grad_norm": 0.20147347450256348,
+ "learning_rate": 2.8572386058981235e-06,
+ "loss": 0.4737790584564209,
+ "step": 6040
+ },
+ {
+ "epoch": 1.6247276870999094,
+ "grad_norm": 0.2455863654613495,
+ "learning_rate": 2.8170241286863272e-06,
+ "loss": 0.4722298145294189,
+ "step": 6060
+ },
+ {
+ "epoch": 1.6300901565170762,
+ "grad_norm": 0.22172148525714874,
+ "learning_rate": 2.776809651474531e-06,
+ "loss": 0.5120372295379638,
+ "step": 6080
+ },
+ {
+ "epoch": 1.6354526259342426,
+ "grad_norm": 0.3848462700843811,
+ "learning_rate": 2.7365951742627346e-06,
+ "loss": 0.5152206897735596,
+ "step": 6100
+ },
+ {
+ "epoch": 1.6408150953514093,
+ "grad_norm": 0.19071047008037567,
+ "learning_rate": 2.6963806970509383e-06,
+ "loss": 0.4757692813873291,
+ "step": 6120
+ },
+ {
+ "epoch": 1.646177564768576,
+ "grad_norm": 0.20568661391735077,
+ "learning_rate": 2.656166219839142e-06,
+ "loss": 0.475917387008667,
+ "step": 6140
+ },
+ {
+ "epoch": 1.6515400341857425,
+ "grad_norm": 0.11777322739362717,
+ "learning_rate": 2.6159517426273457e-06,
+ "loss": 0.5161296367645264,
+ "step": 6160
+ },
+ {
+ "epoch": 1.6569025036029092,
+ "grad_norm": 0.1700555831193924,
+ "learning_rate": 2.5757372654155494e-06,
+ "loss": 0.4715432167053223,
+ "step": 6180
+ },
+ {
+ "epoch": 1.6622649730200756,
+ "grad_norm": 0.18927083909511566,
+ "learning_rate": 2.5355227882037535e-06,
+ "loss": 0.49937710762023924,
+ "step": 6200
+ },
+ {
+ "epoch": 1.6676274424372424,
+ "grad_norm": 0.22097784280776978,
+ "learning_rate": 2.4953083109919572e-06,
+ "loss": 0.43366107940673826,
+ "step": 6220
+ },
+ {
+ "epoch": 1.672989911854409,
+ "grad_norm": 0.2299281805753708,
+ "learning_rate": 2.455093833780161e-06,
+ "loss": 0.5145821094512939,
+ "step": 6240
+ },
+ {
+ "epoch": 1.6783523812715755,
+ "grad_norm": 0.2384844720363617,
+ "learning_rate": 2.4148793565683646e-06,
+ "loss": 0.459308385848999,
+ "step": 6260
+ },
+ {
+ "epoch": 1.6837148506887423,
+ "grad_norm": 0.24471035599708557,
+ "learning_rate": 2.3746648793565683e-06,
+ "loss": 0.4676504611968994,
+ "step": 6280
+ },
+ {
+ "epoch": 1.6890773201059086,
+ "grad_norm": 0.24419866502285004,
+ "learning_rate": 2.334450402144772e-06,
+ "loss": 0.4745138168334961,
+ "step": 6300
+ },
+ {
+ "epoch": 1.6944397895230754,
+ "grad_norm": 0.15896575152873993,
+ "learning_rate": 2.294235924932976e-06,
+ "loss": 0.5073649883270264,
+ "step": 6320
+ },
+ {
+ "epoch": 1.699802258940242,
+ "grad_norm": 0.26504868268966675,
+ "learning_rate": 2.25402144772118e-06,
+ "loss": 0.4534353733062744,
+ "step": 6340
+ },
+ {
+ "epoch": 1.7051647283574085,
+ "grad_norm": 0.2461850792169571,
+ "learning_rate": 2.2138069705093836e-06,
+ "loss": 0.4862947940826416,
+ "step": 6360
+ },
+ {
+ "epoch": 1.7105271977745753,
+ "grad_norm": 0.17332817614078522,
+ "learning_rate": 2.1735924932975873e-06,
+ "loss": 0.5049370765686035,
+ "step": 6380
+ },
+ {
+ "epoch": 1.7158896671917419,
+ "grad_norm": 0.19762548804283142,
+ "learning_rate": 2.133378016085791e-06,
+ "loss": 0.5272616386413574,
+ "step": 6400
+ },
+ {
+ "epoch": 1.7212521366089084,
+ "grad_norm": 0.23265399038791656,
+ "learning_rate": 2.0931635388739946e-06,
+ "loss": 0.47600841522216797,
+ "step": 6420
+ },
+ {
+ "epoch": 1.726614606026075,
+ "grad_norm": 0.20868578553199768,
+ "learning_rate": 2.0529490616621983e-06,
+ "loss": 0.5027226448059082,
+ "step": 6440
+ },
+ {
+ "epoch": 1.7319770754432415,
+ "grad_norm": 0.2851981520652771,
+ "learning_rate": 2.012734584450402e-06,
+ "loss": 0.5288124561309815,
+ "step": 6460
+ },
+ {
+ "epoch": 1.7373395448604083,
+ "grad_norm": 0.20086587965488434,
+ "learning_rate": 1.9725201072386057e-06,
+ "loss": 0.4625516891479492,
+ "step": 6480
+ },
+ {
+ "epoch": 1.7427020142775749,
+ "grad_norm": 0.24060192704200745,
+ "learning_rate": 1.93230563002681e-06,
+ "loss": 0.4843903541564941,
+ "step": 6500
+ },
+ {
+ "epoch": 1.7480644836947414,
+ "grad_norm": 0.33561915159225464,
+ "learning_rate": 1.8920911528150133e-06,
+ "loss": 0.4823720932006836,
+ "step": 6520
+ },
+ {
+ "epoch": 1.753426953111908,
+ "grad_norm": 0.2510465383529663,
+ "learning_rate": 1.851876675603217e-06,
+ "loss": 0.46517143249511717,
+ "step": 6540
+ },
+ {
+ "epoch": 1.7587894225290746,
+ "grad_norm": 0.2631177604198456,
+ "learning_rate": 1.811662198391421e-06,
+ "loss": 0.5004732131958007,
+ "step": 6560
+ },
+ {
+ "epoch": 1.7641518919462413,
+ "grad_norm": 0.3493230640888214,
+ "learning_rate": 1.7714477211796249e-06,
+ "loss": 0.523811674118042,
+ "step": 6580
+ },
+ {
+ "epoch": 1.769514361363408,
+ "grad_norm": 0.1742691546678543,
+ "learning_rate": 1.7312332439678286e-06,
+ "loss": 0.5276295661926269,
+ "step": 6600
+ },
+ {
+ "epoch": 1.7748768307805745,
+ "grad_norm": 0.16134823858737946,
+ "learning_rate": 1.6910187667560323e-06,
+ "loss": 0.5352637290954589,
+ "step": 6620
+ },
+ {
+ "epoch": 1.780239300197741,
+ "grad_norm": 0.20977018773555756,
+ "learning_rate": 1.650804289544236e-06,
+ "loss": 0.4955774784088135,
+ "step": 6640
+ },
+ {
+ "epoch": 1.7856017696149076,
+ "grad_norm": 0.20511005818843842,
+ "learning_rate": 1.6105898123324397e-06,
+ "loss": 0.48643174171447756,
+ "step": 6660
+ },
+ {
+ "epoch": 1.7909642390320744,
+ "grad_norm": 0.23870044946670532,
+ "learning_rate": 1.5703753351206434e-06,
+ "loss": 0.4673162460327148,
+ "step": 6680
+ },
+ {
+ "epoch": 1.796326708449241,
+ "grad_norm": 0.21660065650939941,
+ "learning_rate": 1.5301608579088473e-06,
+ "loss": 0.5381903648376465,
+ "step": 6700
+ },
+ {
+ "epoch": 1.8016891778664075,
+ "grad_norm": 0.26977139711380005,
+ "learning_rate": 1.489946380697051e-06,
+ "loss": 0.42094998359680175,
+ "step": 6720
+ },
+ {
+ "epoch": 1.807051647283574,
+ "grad_norm": 0.2088550478219986,
+ "learning_rate": 1.4497319034852549e-06,
+ "loss": 0.49211792945861815,
+ "step": 6740
+ },
+ {
+ "epoch": 1.8124141167007406,
+ "grad_norm": 0.18141885101795197,
+ "learning_rate": 1.4095174262734586e-06,
+ "loss": 0.46572179794311525,
+ "step": 6760
+ },
+ {
+ "epoch": 1.8177765861179074,
+ "grad_norm": 0.2200685739517212,
+ "learning_rate": 1.3693029490616623e-06,
+ "loss": 0.4996177196502686,
+ "step": 6780
+ },
+ {
+ "epoch": 1.823139055535074,
+ "grad_norm": 0.19545452296733856,
+ "learning_rate": 1.329088471849866e-06,
+ "loss": 0.4731945514678955,
+ "step": 6800
+ },
+ {
+ "epoch": 1.8285015249522405,
+ "grad_norm": 0.2239731252193451,
+ "learning_rate": 1.2888739946380697e-06,
+ "loss": 0.47544050216674805,
+ "step": 6820
+ },
+ {
+ "epoch": 1.833863994369407,
+ "grad_norm": 0.22336581349372864,
+ "learning_rate": 1.2486595174262734e-06,
+ "loss": 0.47878737449645997,
+ "step": 6840
+ },
+ {
+ "epoch": 1.8392264637865736,
+ "grad_norm": 0.20921571552753448,
+ "learning_rate": 1.2084450402144773e-06,
+ "loss": 0.41347403526306153,
+ "step": 6860
+ },
+ {
+ "epoch": 1.8445889332037404,
+ "grad_norm": 0.1577194333076477,
+ "learning_rate": 1.168230563002681e-06,
+ "loss": 0.5464958667755127,
+ "step": 6880
+ },
+ {
+ "epoch": 1.849951402620907,
+ "grad_norm": 0.1477355808019638,
+ "learning_rate": 1.1280160857908849e-06,
+ "loss": 0.48548617362976076,
+ "step": 6900
+ },
+ {
+ "epoch": 1.8553138720380735,
+ "grad_norm": 0.22352682054042816,
+ "learning_rate": 1.0878016085790886e-06,
+ "loss": 0.4518588542938232,
+ "step": 6920
+ },
+ {
+ "epoch": 1.8606763414552403,
+ "grad_norm": 0.19822706282138824,
+ "learning_rate": 1.0475871313672923e-06,
+ "loss": 0.4190972805023193,
+ "step": 6940
+ },
+ {
+ "epoch": 1.8660388108724066,
+ "grad_norm": 0.20670010149478912,
+ "learning_rate": 1.007372654155496e-06,
+ "loss": 0.5045090675354004,
+ "step": 6960
+ },
+ {
+ "epoch": 1.8714012802895734,
+ "grad_norm": 0.2154514342546463,
+ "learning_rate": 9.671581769436997e-07,
+ "loss": 0.4472982883453369,
+ "step": 6980
+ },
+ {
+ "epoch": 1.87676374970674,
+ "grad_norm": 0.19451302289962769,
+ "learning_rate": 9.269436997319035e-07,
+ "loss": 0.4328409194946289,
+ "step": 7000
+ },
+ {
+ "epoch": 1.8821262191239065,
+ "grad_norm": 0.20980985462665558,
+ "learning_rate": 8.867292225201073e-07,
+ "loss": 0.43312845230102537,
+ "step": 7020
+ },
+ {
+ "epoch": 1.8874886885410733,
+ "grad_norm": 0.19927652180194855,
+ "learning_rate": 8.46514745308311e-07,
+ "loss": 0.5255829811096191,
+ "step": 7040
+ },
+ {
+ "epoch": 1.8928511579582397,
+ "grad_norm": 0.2869090437889099,
+ "learning_rate": 8.063002680965148e-07,
+ "loss": 0.5301108360290527,
+ "step": 7060
+ },
+ {
+ "epoch": 1.8982136273754064,
+ "grad_norm": 0.16662724316120148,
+ "learning_rate": 7.660857908847185e-07,
+ "loss": 0.48042588233947753,
+ "step": 7080
+ },
+ {
+ "epoch": 1.903576096792573,
+ "grad_norm": 0.21045279502868652,
+ "learning_rate": 7.258713136729223e-07,
+ "loss": 0.4943391799926758,
+ "step": 7100
+ },
+ {
+ "epoch": 1.9089385662097396,
+ "grad_norm": 0.18638195097446442,
+ "learning_rate": 6.85656836461126e-07,
+ "loss": 0.48406500816345216,
+ "step": 7120
+ },
+ {
+ "epoch": 1.9143010356269063,
+ "grad_norm": 0.15428832173347473,
+ "learning_rate": 6.454423592493298e-07,
+ "loss": 0.5084923267364502,
+ "step": 7140
+ },
+ {
+ "epoch": 1.9196635050440727,
+ "grad_norm": 0.2415294051170349,
+ "learning_rate": 6.052278820375336e-07,
+ "loss": 0.5026498317718506,
+ "step": 7160
+ },
+ {
+ "epoch": 1.9250259744612395,
+ "grad_norm": 0.23021087050437927,
+ "learning_rate": 5.650134048257373e-07,
+ "loss": 0.5294596195220947,
+ "step": 7180
+ },
+ {
+ "epoch": 1.930388443878406,
+ "grad_norm": 0.21689893305301666,
+ "learning_rate": 5.24798927613941e-07,
+ "loss": 0.4569683074951172,
+ "step": 7200
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 8.868448234493706e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-7200/training_args.bin b/checkpoint-7200/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-7200/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-7400/README.md b/checkpoint-7400/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-7400/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-7400/adapter_config.json b/checkpoint-7400/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-7400/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-7400/adapter_model.safetensors b/checkpoint-7400/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..2ac0c41fcf74f204ac49ca5afd4cf72f8acdca3b
--- /dev/null
+++ b/checkpoint-7400/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:abc704f30a112227b3ce512e37c72893b188051a382e1329a88f043367b5ae7c
+size 50360752
diff --git a/checkpoint-7400/chat_template.jinja b/checkpoint-7400/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-7400/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-7400/optimizer.pt b/checkpoint-7400/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..73ed6723d31a896540c4077a7a1c390649d6ad1b
--- /dev/null
+++ b/checkpoint-7400/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a32478a2d44562e34964ded226b738d22fefd30abb28b52a1e2912131a197520
+size 100828235
diff --git a/checkpoint-7400/rng_state.pth b/checkpoint-7400/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..a3535d199a9425419fc2a0776faa98eea7cf24e0
--- /dev/null
+++ b/checkpoint-7400/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:80bb9c70314145190b1bed112a68c6d20e2cf4d8364307a86eaaec3f257f16ad
+size 14645
diff --git a/checkpoint-7400/scheduler.pt b/checkpoint-7400/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..c3048c9532c6d2653b4ea9c2c088faf5054997c1
--- /dev/null
+++ b/checkpoint-7400/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:00c1d3473202ef174820e054d21dcd57620585a9ad949f4a2124d723b5c9572f
+size 1465
diff --git a/checkpoint-7400/tokenizer.json b/checkpoint-7400/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-7400/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-7400/tokenizer_config.json b/checkpoint-7400/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-7400/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-7400/trainer_state.json b/checkpoint-7400/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..dbdf241193f77d5519a0fd8883f0f1b62f13646a
--- /dev/null
+++ b/checkpoint-7400/trainer_state.json
@@ -0,0 +1,2624 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 1.984013138050072,
+ "eval_steps": 500,
+ "global_step": 7400,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ },
+ {
+ "epoch": 1.185005194892248,
+ "grad_norm": 0.16667844355106354,
+ "learning_rate": 6.114611260053619e-06,
+ "loss": 0.5159809589385986,
+ "step": 4420
+ },
+ {
+ "epoch": 1.1903676643094145,
+ "grad_norm": 0.19206570088863373,
+ "learning_rate": 6.074396782841823e-06,
+ "loss": 0.47541089057922364,
+ "step": 4440
+ },
+ {
+ "epoch": 1.195730133726581,
+ "grad_norm": 0.17394617199897766,
+ "learning_rate": 6.034182305630027e-06,
+ "loss": 0.5470661640167236,
+ "step": 4460
+ },
+ {
+ "epoch": 1.2010926031437477,
+ "grad_norm": 0.210404634475708,
+ "learning_rate": 5.993967828418231e-06,
+ "loss": 0.5377882957458496,
+ "step": 4480
+ },
+ {
+ "epoch": 1.2064550725609142,
+ "grad_norm": 0.18084648251533508,
+ "learning_rate": 5.953753351206435e-06,
+ "loss": 0.5037185192108155,
+ "step": 4500
+ },
+ {
+ "epoch": 1.211817541978081,
+ "grad_norm": 0.23707027733325958,
+ "learning_rate": 5.913538873994639e-06,
+ "loss": 0.4822190284729004,
+ "step": 4520
+ },
+ {
+ "epoch": 1.2171800113952476,
+ "grad_norm": 0.16474473476409912,
+ "learning_rate": 5.873324396782842e-06,
+ "loss": 0.46645288467407225,
+ "step": 4540
+ },
+ {
+ "epoch": 1.2225424808124141,
+ "grad_norm": 0.2142348438501358,
+ "learning_rate": 5.833109919571046e-06,
+ "loss": 0.5255855560302735,
+ "step": 4560
+ },
+ {
+ "epoch": 1.2279049502295807,
+ "grad_norm": 0.2531765103340149,
+ "learning_rate": 5.79289544235925e-06,
+ "loss": 0.507044792175293,
+ "step": 4580
+ },
+ {
+ "epoch": 1.2332674196467472,
+ "grad_norm": 0.2553550899028778,
+ "learning_rate": 5.7526809651474535e-06,
+ "loss": 0.4767824649810791,
+ "step": 4600
+ },
+ {
+ "epoch": 1.238629889063914,
+ "grad_norm": 0.14484412968158722,
+ "learning_rate": 5.712466487935657e-06,
+ "loss": 0.4675601005554199,
+ "step": 4620
+ },
+ {
+ "epoch": 1.2439923584810806,
+ "grad_norm": 0.14328251779079437,
+ "learning_rate": 5.672252010723861e-06,
+ "loss": 0.4956005573272705,
+ "step": 4640
+ },
+ {
+ "epoch": 1.2493548278982471,
+ "grad_norm": 0.1739245355129242,
+ "learning_rate": 5.632037533512065e-06,
+ "loss": 0.48583345413208007,
+ "step": 4660
+ },
+ {
+ "epoch": 1.2547172973154137,
+ "grad_norm": 0.21294184029102325,
+ "learning_rate": 5.591823056300268e-06,
+ "loss": 0.520921277999878,
+ "step": 4680
+ },
+ {
+ "epoch": 1.2600797667325803,
+ "grad_norm": 0.25132355093955994,
+ "learning_rate": 5.551608579088472e-06,
+ "loss": 0.5295385837554931,
+ "step": 4700
+ },
+ {
+ "epoch": 1.265442236149747,
+ "grad_norm": 0.18603841960430145,
+ "learning_rate": 5.511394101876676e-06,
+ "loss": 0.47570199966430665,
+ "step": 4720
+ },
+ {
+ "epoch": 1.2708047055669136,
+ "grad_norm": 0.19883134961128235,
+ "learning_rate": 5.471179624664879e-06,
+ "loss": 0.5016080379486084,
+ "step": 4740
+ },
+ {
+ "epoch": 1.2761671749840802,
+ "grad_norm": 0.19640181958675385,
+ "learning_rate": 5.430965147453083e-06,
+ "loss": 0.4999081134796143,
+ "step": 4760
+ },
+ {
+ "epoch": 1.2815296444012467,
+ "grad_norm": 0.2584764361381531,
+ "learning_rate": 5.390750670241287e-06,
+ "loss": 0.4780082702636719,
+ "step": 4780
+ },
+ {
+ "epoch": 1.2868921138184133,
+ "grad_norm": 0.2925741374492645,
+ "learning_rate": 5.3505361930294905e-06,
+ "loss": 0.5131395816802978,
+ "step": 4800
+ },
+ {
+ "epoch": 1.29225458323558,
+ "grad_norm": 0.18971531093120575,
+ "learning_rate": 5.310321715817694e-06,
+ "loss": 0.455674409866333,
+ "step": 4820
+ },
+ {
+ "epoch": 1.2976170526527466,
+ "grad_norm": 0.16778405010700226,
+ "learning_rate": 5.270107238605898e-06,
+ "loss": 0.5070962905883789,
+ "step": 4840
+ },
+ {
+ "epoch": 1.3029795220699132,
+ "grad_norm": 0.30026957392692566,
+ "learning_rate": 5.2298927613941016e-06,
+ "loss": 0.5120027542114258,
+ "step": 4860
+ },
+ {
+ "epoch": 1.3083419914870797,
+ "grad_norm": 0.17846634984016418,
+ "learning_rate": 5.189678284182305e-06,
+ "loss": 0.5114477157592774,
+ "step": 4880
+ },
+ {
+ "epoch": 1.3137044609042463,
+ "grad_norm": 0.1962418258190155,
+ "learning_rate": 5.149463806970509e-06,
+ "loss": 0.5043613910675049,
+ "step": 4900
+ },
+ {
+ "epoch": 1.319066930321413,
+ "grad_norm": 0.18446756899356842,
+ "learning_rate": 5.1092493297587135e-06,
+ "loss": 0.5396455287933349,
+ "step": 4920
+ },
+ {
+ "epoch": 1.3244293997385796,
+ "grad_norm": 0.20886844396591187,
+ "learning_rate": 5.069034852546917e-06,
+ "loss": 0.4879767417907715,
+ "step": 4940
+ },
+ {
+ "epoch": 1.3297918691557462,
+ "grad_norm": 0.16687901318073273,
+ "learning_rate": 5.028820375335121e-06,
+ "loss": 0.5014327049255372,
+ "step": 4960
+ },
+ {
+ "epoch": 1.3351543385729128,
+ "grad_norm": 0.19595153629779816,
+ "learning_rate": 4.988605898123325e-06,
+ "loss": 0.5375277996063232,
+ "step": 4980
+ },
+ {
+ "epoch": 1.3405168079900793,
+ "grad_norm": 0.2372344732284546,
+ "learning_rate": 4.948391420911528e-06,
+ "loss": 0.5020076274871826,
+ "step": 5000
+ },
+ {
+ "epoch": 1.345879277407246,
+ "grad_norm": 0.21030014753341675,
+ "learning_rate": 4.908176943699732e-06,
+ "loss": 0.5111066818237304,
+ "step": 5020
+ },
+ {
+ "epoch": 1.3512417468244127,
+ "grad_norm": 0.1866692751646042,
+ "learning_rate": 4.867962466487936e-06,
+ "loss": 0.4515383720397949,
+ "step": 5040
+ },
+ {
+ "epoch": 1.3566042162415792,
+ "grad_norm": 0.22531798481941223,
+ "learning_rate": 4.827747989276139e-06,
+ "loss": 0.4757690906524658,
+ "step": 5060
+ },
+ {
+ "epoch": 1.3619666856587458,
+ "grad_norm": 0.15868768095970154,
+ "learning_rate": 4.787533512064343e-06,
+ "loss": 0.45842318534851073,
+ "step": 5080
+ },
+ {
+ "epoch": 1.3673291550759124,
+ "grad_norm": 0.24528546631336212,
+ "learning_rate": 4.747319034852547e-06,
+ "loss": 0.47269258499145506,
+ "step": 5100
+ },
+ {
+ "epoch": 1.3726916244930791,
+ "grad_norm": 0.17387732863426208,
+ "learning_rate": 4.707104557640751e-06,
+ "loss": 0.5103805065155029,
+ "step": 5120
+ },
+ {
+ "epoch": 1.3780540939102457,
+ "grad_norm": 0.20686905086040497,
+ "learning_rate": 4.666890080428955e-06,
+ "loss": 0.5135180950164795,
+ "step": 5140
+ },
+ {
+ "epoch": 1.3834165633274123,
+ "grad_norm": 0.19599783420562744,
+ "learning_rate": 4.626675603217159e-06,
+ "loss": 0.5045839786529541,
+ "step": 5160
+ },
+ {
+ "epoch": 1.3887790327445788,
+ "grad_norm": 0.2585010528564453,
+ "learning_rate": 4.586461126005362e-06,
+ "loss": 0.45903496742248534,
+ "step": 5180
+ },
+ {
+ "epoch": 1.3941415021617454,
+ "grad_norm": 0.1688319593667984,
+ "learning_rate": 4.546246648793566e-06,
+ "loss": 0.5017509937286377,
+ "step": 5200
+ },
+ {
+ "epoch": 1.3995039715789122,
+ "grad_norm": 0.21520815789699554,
+ "learning_rate": 4.50603217158177e-06,
+ "loss": 0.48459539413452146,
+ "step": 5220
+ },
+ {
+ "epoch": 1.4048664409960787,
+ "grad_norm": 0.20514647662639618,
+ "learning_rate": 4.4658176943699735e-06,
+ "loss": 0.5073423862457276,
+ "step": 5240
+ },
+ {
+ "epoch": 1.4102289104132453,
+ "grad_norm": 0.21835413575172424,
+ "learning_rate": 4.425603217158177e-06,
+ "loss": 0.5290310382843018,
+ "step": 5260
+ },
+ {
+ "epoch": 1.4155913798304118,
+ "grad_norm": 0.28042587637901306,
+ "learning_rate": 4.385388739946381e-06,
+ "loss": 0.4823312759399414,
+ "step": 5280
+ },
+ {
+ "epoch": 1.4209538492475784,
+ "grad_norm": 0.18959026038646698,
+ "learning_rate": 4.345174262734585e-06,
+ "loss": 0.4921241760253906,
+ "step": 5300
+ },
+ {
+ "epoch": 1.4263163186647452,
+ "grad_norm": 0.18584316968917847,
+ "learning_rate": 4.304959785522788e-06,
+ "loss": 0.4892130374908447,
+ "step": 5320
+ },
+ {
+ "epoch": 1.4316787880819117,
+ "grad_norm": 0.17588038742542267,
+ "learning_rate": 4.264745308310992e-06,
+ "loss": 0.4822041988372803,
+ "step": 5340
+ },
+ {
+ "epoch": 1.4370412574990783,
+ "grad_norm": 0.18146033585071564,
+ "learning_rate": 4.224530831099196e-06,
+ "loss": 0.5084807395935058,
+ "step": 5360
+ },
+ {
+ "epoch": 1.4424037269162449,
+ "grad_norm": 0.2251797467470169,
+ "learning_rate": 4.184316353887399e-06,
+ "loss": 0.5146170139312745,
+ "step": 5380
+ },
+ {
+ "epoch": 1.4477661963334114,
+ "grad_norm": 0.18744796514511108,
+ "learning_rate": 4.144101876675603e-06,
+ "loss": 0.5189927577972412,
+ "step": 5400
+ },
+ {
+ "epoch": 1.4531286657505782,
+ "grad_norm": 0.25737133622169495,
+ "learning_rate": 4.103887399463807e-06,
+ "loss": 0.4891658782958984,
+ "step": 5420
+ },
+ {
+ "epoch": 1.4584911351677448,
+ "grad_norm": 0.20580479502677917,
+ "learning_rate": 4.0636729222520105e-06,
+ "loss": 0.4953591823577881,
+ "step": 5440
+ },
+ {
+ "epoch": 1.4638536045849113,
+ "grad_norm": 0.2351546287536621,
+ "learning_rate": 4.023458445040214e-06,
+ "loss": 0.5025320053100586,
+ "step": 5460
+ },
+ {
+ "epoch": 1.4692160740020779,
+ "grad_norm": 0.1819481998682022,
+ "learning_rate": 3.983243967828418e-06,
+ "loss": 0.47151756286621094,
+ "step": 5480
+ },
+ {
+ "epoch": 1.4745785434192444,
+ "grad_norm": 0.20772472023963928,
+ "learning_rate": 3.943029490616622e-06,
+ "loss": 0.4678915023803711,
+ "step": 5500
+ },
+ {
+ "epoch": 1.4799410128364112,
+ "grad_norm": 0.2203037440776825,
+ "learning_rate": 3.902815013404825e-06,
+ "loss": 0.46007452011108396,
+ "step": 5520
+ },
+ {
+ "epoch": 1.4853034822535778,
+ "grad_norm": 0.15371400117874146,
+ "learning_rate": 3.86260053619303e-06,
+ "loss": 0.44407024383544924,
+ "step": 5540
+ },
+ {
+ "epoch": 1.4906659516707443,
+ "grad_norm": 0.2276080846786499,
+ "learning_rate": 3.8223860589812335e-06,
+ "loss": 0.4730556488037109,
+ "step": 5560
+ },
+ {
+ "epoch": 1.4960284210879111,
+ "grad_norm": 0.24482466280460358,
+ "learning_rate": 3.7821715817694376e-06,
+ "loss": 0.5073911666870117,
+ "step": 5580
+ },
+ {
+ "epoch": 1.5013908905050775,
+ "grad_norm": 0.20438458025455475,
+ "learning_rate": 3.741957104557641e-06,
+ "loss": 0.46701641082763673,
+ "step": 5600
+ },
+ {
+ "epoch": 1.5067533599222442,
+ "grad_norm": 0.19854313135147095,
+ "learning_rate": 3.7017426273458446e-06,
+ "loss": 0.46309399604797363,
+ "step": 5620
+ },
+ {
+ "epoch": 1.5121158293394108,
+ "grad_norm": 0.18356069922447205,
+ "learning_rate": 3.6615281501340483e-06,
+ "loss": 0.503613805770874,
+ "step": 5640
+ },
+ {
+ "epoch": 1.5174782987565774,
+ "grad_norm": 0.2009744495153427,
+ "learning_rate": 3.621313672922252e-06,
+ "loss": 0.4765054225921631,
+ "step": 5660
+ },
+ {
+ "epoch": 1.5228407681737441,
+ "grad_norm": 0.3058745563030243,
+ "learning_rate": 3.5810991957104557e-06,
+ "loss": 0.5179148197174073,
+ "step": 5680
+ },
+ {
+ "epoch": 1.5282032375909105,
+ "grad_norm": 0.17671597003936768,
+ "learning_rate": 3.54088471849866e-06,
+ "loss": 0.45907344818115237,
+ "step": 5700
+ },
+ {
+ "epoch": 1.5335657070080773,
+ "grad_norm": 0.22209160029888153,
+ "learning_rate": 3.5006702412868635e-06,
+ "loss": 0.49304862022399903,
+ "step": 5720
+ },
+ {
+ "epoch": 1.5389281764252438,
+ "grad_norm": 0.21018914878368378,
+ "learning_rate": 3.4604557640750672e-06,
+ "loss": 0.5536758422851562,
+ "step": 5740
+ },
+ {
+ "epoch": 1.5442906458424104,
+ "grad_norm": 0.14339996874332428,
+ "learning_rate": 3.420241286863271e-06,
+ "loss": 0.48726091384887693,
+ "step": 5760
+ },
+ {
+ "epoch": 1.5496531152595772,
+ "grad_norm": 0.11419746279716492,
+ "learning_rate": 3.3800268096514746e-06,
+ "loss": 0.4514151573181152,
+ "step": 5780
+ },
+ {
+ "epoch": 1.5550155846767435,
+ "grad_norm": 0.18168962001800537,
+ "learning_rate": 3.3398123324396783e-06,
+ "loss": 0.5279990196228027,
+ "step": 5800
+ },
+ {
+ "epoch": 1.5603780540939103,
+ "grad_norm": 0.24244488775730133,
+ "learning_rate": 3.299597855227882e-06,
+ "loss": 0.49297361373901366,
+ "step": 5820
+ },
+ {
+ "epoch": 1.5657405235110768,
+ "grad_norm": 0.2017296999692917,
+ "learning_rate": 3.2593833780160857e-06,
+ "loss": 0.49305019378662107,
+ "step": 5840
+ },
+ {
+ "epoch": 1.5711029929282434,
+ "grad_norm": 0.22592377662658691,
+ "learning_rate": 3.2191689008042894e-06,
+ "loss": 0.4862989902496338,
+ "step": 5860
+ },
+ {
+ "epoch": 1.5764654623454102,
+ "grad_norm": 0.24772357940673828,
+ "learning_rate": 3.1789544235924935e-06,
+ "loss": 0.45182647705078127,
+ "step": 5880
+ },
+ {
+ "epoch": 1.5818279317625765,
+ "grad_norm": 0.20607218146324158,
+ "learning_rate": 3.1387399463806972e-06,
+ "loss": 0.48905248641967775,
+ "step": 5900
+ },
+ {
+ "epoch": 1.5871904011797433,
+ "grad_norm": 0.1931353509426117,
+ "learning_rate": 3.098525469168901e-06,
+ "loss": 0.5307461261749268,
+ "step": 5920
+ },
+ {
+ "epoch": 1.5925528705969099,
+ "grad_norm": 0.16020581126213074,
+ "learning_rate": 3.0583109919571046e-06,
+ "loss": 0.4672811985015869,
+ "step": 5940
+ },
+ {
+ "epoch": 1.5979153400140764,
+ "grad_norm": 0.23668015003204346,
+ "learning_rate": 3.0180965147453083e-06,
+ "loss": 0.5272688865661621,
+ "step": 5960
+ },
+ {
+ "epoch": 1.6032778094312432,
+ "grad_norm": 0.1916576772928238,
+ "learning_rate": 2.977882037533512e-06,
+ "loss": 0.4859332084655762,
+ "step": 5980
+ },
+ {
+ "epoch": 1.6086402788484095,
+ "grad_norm": 0.23635101318359375,
+ "learning_rate": 2.9376675603217157e-06,
+ "loss": 0.5418910980224609,
+ "step": 6000
+ },
+ {
+ "epoch": 1.6140027482655763,
+ "grad_norm": 0.2404562532901764,
+ "learning_rate": 2.89745308310992e-06,
+ "loss": 0.5449445247650146,
+ "step": 6020
+ },
+ {
+ "epoch": 1.6193652176827429,
+ "grad_norm": 0.20147347450256348,
+ "learning_rate": 2.8572386058981235e-06,
+ "loss": 0.4737790584564209,
+ "step": 6040
+ },
+ {
+ "epoch": 1.6247276870999094,
+ "grad_norm": 0.2455863654613495,
+ "learning_rate": 2.8170241286863272e-06,
+ "loss": 0.4722298145294189,
+ "step": 6060
+ },
+ {
+ "epoch": 1.6300901565170762,
+ "grad_norm": 0.22172148525714874,
+ "learning_rate": 2.776809651474531e-06,
+ "loss": 0.5120372295379638,
+ "step": 6080
+ },
+ {
+ "epoch": 1.6354526259342426,
+ "grad_norm": 0.3848462700843811,
+ "learning_rate": 2.7365951742627346e-06,
+ "loss": 0.5152206897735596,
+ "step": 6100
+ },
+ {
+ "epoch": 1.6408150953514093,
+ "grad_norm": 0.19071047008037567,
+ "learning_rate": 2.6963806970509383e-06,
+ "loss": 0.4757692813873291,
+ "step": 6120
+ },
+ {
+ "epoch": 1.646177564768576,
+ "grad_norm": 0.20568661391735077,
+ "learning_rate": 2.656166219839142e-06,
+ "loss": 0.475917387008667,
+ "step": 6140
+ },
+ {
+ "epoch": 1.6515400341857425,
+ "grad_norm": 0.11777322739362717,
+ "learning_rate": 2.6159517426273457e-06,
+ "loss": 0.5161296367645264,
+ "step": 6160
+ },
+ {
+ "epoch": 1.6569025036029092,
+ "grad_norm": 0.1700555831193924,
+ "learning_rate": 2.5757372654155494e-06,
+ "loss": 0.4715432167053223,
+ "step": 6180
+ },
+ {
+ "epoch": 1.6622649730200756,
+ "grad_norm": 0.18927083909511566,
+ "learning_rate": 2.5355227882037535e-06,
+ "loss": 0.49937710762023924,
+ "step": 6200
+ },
+ {
+ "epoch": 1.6676274424372424,
+ "grad_norm": 0.22097784280776978,
+ "learning_rate": 2.4953083109919572e-06,
+ "loss": 0.43366107940673826,
+ "step": 6220
+ },
+ {
+ "epoch": 1.672989911854409,
+ "grad_norm": 0.2299281805753708,
+ "learning_rate": 2.455093833780161e-06,
+ "loss": 0.5145821094512939,
+ "step": 6240
+ },
+ {
+ "epoch": 1.6783523812715755,
+ "grad_norm": 0.2384844720363617,
+ "learning_rate": 2.4148793565683646e-06,
+ "loss": 0.459308385848999,
+ "step": 6260
+ },
+ {
+ "epoch": 1.6837148506887423,
+ "grad_norm": 0.24471035599708557,
+ "learning_rate": 2.3746648793565683e-06,
+ "loss": 0.4676504611968994,
+ "step": 6280
+ },
+ {
+ "epoch": 1.6890773201059086,
+ "grad_norm": 0.24419866502285004,
+ "learning_rate": 2.334450402144772e-06,
+ "loss": 0.4745138168334961,
+ "step": 6300
+ },
+ {
+ "epoch": 1.6944397895230754,
+ "grad_norm": 0.15896575152873993,
+ "learning_rate": 2.294235924932976e-06,
+ "loss": 0.5073649883270264,
+ "step": 6320
+ },
+ {
+ "epoch": 1.699802258940242,
+ "grad_norm": 0.26504868268966675,
+ "learning_rate": 2.25402144772118e-06,
+ "loss": 0.4534353733062744,
+ "step": 6340
+ },
+ {
+ "epoch": 1.7051647283574085,
+ "grad_norm": 0.2461850792169571,
+ "learning_rate": 2.2138069705093836e-06,
+ "loss": 0.4862947940826416,
+ "step": 6360
+ },
+ {
+ "epoch": 1.7105271977745753,
+ "grad_norm": 0.17332817614078522,
+ "learning_rate": 2.1735924932975873e-06,
+ "loss": 0.5049370765686035,
+ "step": 6380
+ },
+ {
+ "epoch": 1.7158896671917419,
+ "grad_norm": 0.19762548804283142,
+ "learning_rate": 2.133378016085791e-06,
+ "loss": 0.5272616386413574,
+ "step": 6400
+ },
+ {
+ "epoch": 1.7212521366089084,
+ "grad_norm": 0.23265399038791656,
+ "learning_rate": 2.0931635388739946e-06,
+ "loss": 0.47600841522216797,
+ "step": 6420
+ },
+ {
+ "epoch": 1.726614606026075,
+ "grad_norm": 0.20868578553199768,
+ "learning_rate": 2.0529490616621983e-06,
+ "loss": 0.5027226448059082,
+ "step": 6440
+ },
+ {
+ "epoch": 1.7319770754432415,
+ "grad_norm": 0.2851981520652771,
+ "learning_rate": 2.012734584450402e-06,
+ "loss": 0.5288124561309815,
+ "step": 6460
+ },
+ {
+ "epoch": 1.7373395448604083,
+ "grad_norm": 0.20086587965488434,
+ "learning_rate": 1.9725201072386057e-06,
+ "loss": 0.4625516891479492,
+ "step": 6480
+ },
+ {
+ "epoch": 1.7427020142775749,
+ "grad_norm": 0.24060192704200745,
+ "learning_rate": 1.93230563002681e-06,
+ "loss": 0.4843903541564941,
+ "step": 6500
+ },
+ {
+ "epoch": 1.7480644836947414,
+ "grad_norm": 0.33561915159225464,
+ "learning_rate": 1.8920911528150133e-06,
+ "loss": 0.4823720932006836,
+ "step": 6520
+ },
+ {
+ "epoch": 1.753426953111908,
+ "grad_norm": 0.2510465383529663,
+ "learning_rate": 1.851876675603217e-06,
+ "loss": 0.46517143249511717,
+ "step": 6540
+ },
+ {
+ "epoch": 1.7587894225290746,
+ "grad_norm": 0.2631177604198456,
+ "learning_rate": 1.811662198391421e-06,
+ "loss": 0.5004732131958007,
+ "step": 6560
+ },
+ {
+ "epoch": 1.7641518919462413,
+ "grad_norm": 0.3493230640888214,
+ "learning_rate": 1.7714477211796249e-06,
+ "loss": 0.523811674118042,
+ "step": 6580
+ },
+ {
+ "epoch": 1.769514361363408,
+ "grad_norm": 0.1742691546678543,
+ "learning_rate": 1.7312332439678286e-06,
+ "loss": 0.5276295661926269,
+ "step": 6600
+ },
+ {
+ "epoch": 1.7748768307805745,
+ "grad_norm": 0.16134823858737946,
+ "learning_rate": 1.6910187667560323e-06,
+ "loss": 0.5352637290954589,
+ "step": 6620
+ },
+ {
+ "epoch": 1.780239300197741,
+ "grad_norm": 0.20977018773555756,
+ "learning_rate": 1.650804289544236e-06,
+ "loss": 0.4955774784088135,
+ "step": 6640
+ },
+ {
+ "epoch": 1.7856017696149076,
+ "grad_norm": 0.20511005818843842,
+ "learning_rate": 1.6105898123324397e-06,
+ "loss": 0.48643174171447756,
+ "step": 6660
+ },
+ {
+ "epoch": 1.7909642390320744,
+ "grad_norm": 0.23870044946670532,
+ "learning_rate": 1.5703753351206434e-06,
+ "loss": 0.4673162460327148,
+ "step": 6680
+ },
+ {
+ "epoch": 1.796326708449241,
+ "grad_norm": 0.21660065650939941,
+ "learning_rate": 1.5301608579088473e-06,
+ "loss": 0.5381903648376465,
+ "step": 6700
+ },
+ {
+ "epoch": 1.8016891778664075,
+ "grad_norm": 0.26977139711380005,
+ "learning_rate": 1.489946380697051e-06,
+ "loss": 0.42094998359680175,
+ "step": 6720
+ },
+ {
+ "epoch": 1.807051647283574,
+ "grad_norm": 0.2088550478219986,
+ "learning_rate": 1.4497319034852549e-06,
+ "loss": 0.49211792945861815,
+ "step": 6740
+ },
+ {
+ "epoch": 1.8124141167007406,
+ "grad_norm": 0.18141885101795197,
+ "learning_rate": 1.4095174262734586e-06,
+ "loss": 0.46572179794311525,
+ "step": 6760
+ },
+ {
+ "epoch": 1.8177765861179074,
+ "grad_norm": 0.2200685739517212,
+ "learning_rate": 1.3693029490616623e-06,
+ "loss": 0.4996177196502686,
+ "step": 6780
+ },
+ {
+ "epoch": 1.823139055535074,
+ "grad_norm": 0.19545452296733856,
+ "learning_rate": 1.329088471849866e-06,
+ "loss": 0.4731945514678955,
+ "step": 6800
+ },
+ {
+ "epoch": 1.8285015249522405,
+ "grad_norm": 0.2239731252193451,
+ "learning_rate": 1.2888739946380697e-06,
+ "loss": 0.47544050216674805,
+ "step": 6820
+ },
+ {
+ "epoch": 1.833863994369407,
+ "grad_norm": 0.22336581349372864,
+ "learning_rate": 1.2486595174262734e-06,
+ "loss": 0.47878737449645997,
+ "step": 6840
+ },
+ {
+ "epoch": 1.8392264637865736,
+ "grad_norm": 0.20921571552753448,
+ "learning_rate": 1.2084450402144773e-06,
+ "loss": 0.41347403526306153,
+ "step": 6860
+ },
+ {
+ "epoch": 1.8445889332037404,
+ "grad_norm": 0.1577194333076477,
+ "learning_rate": 1.168230563002681e-06,
+ "loss": 0.5464958667755127,
+ "step": 6880
+ },
+ {
+ "epoch": 1.849951402620907,
+ "grad_norm": 0.1477355808019638,
+ "learning_rate": 1.1280160857908849e-06,
+ "loss": 0.48548617362976076,
+ "step": 6900
+ },
+ {
+ "epoch": 1.8553138720380735,
+ "grad_norm": 0.22352682054042816,
+ "learning_rate": 1.0878016085790886e-06,
+ "loss": 0.4518588542938232,
+ "step": 6920
+ },
+ {
+ "epoch": 1.8606763414552403,
+ "grad_norm": 0.19822706282138824,
+ "learning_rate": 1.0475871313672923e-06,
+ "loss": 0.4190972805023193,
+ "step": 6940
+ },
+ {
+ "epoch": 1.8660388108724066,
+ "grad_norm": 0.20670010149478912,
+ "learning_rate": 1.007372654155496e-06,
+ "loss": 0.5045090675354004,
+ "step": 6960
+ },
+ {
+ "epoch": 1.8714012802895734,
+ "grad_norm": 0.2154514342546463,
+ "learning_rate": 9.671581769436997e-07,
+ "loss": 0.4472982883453369,
+ "step": 6980
+ },
+ {
+ "epoch": 1.87676374970674,
+ "grad_norm": 0.19451302289962769,
+ "learning_rate": 9.269436997319035e-07,
+ "loss": 0.4328409194946289,
+ "step": 7000
+ },
+ {
+ "epoch": 1.8821262191239065,
+ "grad_norm": 0.20980985462665558,
+ "learning_rate": 8.867292225201073e-07,
+ "loss": 0.43312845230102537,
+ "step": 7020
+ },
+ {
+ "epoch": 1.8874886885410733,
+ "grad_norm": 0.19927652180194855,
+ "learning_rate": 8.46514745308311e-07,
+ "loss": 0.5255829811096191,
+ "step": 7040
+ },
+ {
+ "epoch": 1.8928511579582397,
+ "grad_norm": 0.2869090437889099,
+ "learning_rate": 8.063002680965148e-07,
+ "loss": 0.5301108360290527,
+ "step": 7060
+ },
+ {
+ "epoch": 1.8982136273754064,
+ "grad_norm": 0.16662724316120148,
+ "learning_rate": 7.660857908847185e-07,
+ "loss": 0.48042588233947753,
+ "step": 7080
+ },
+ {
+ "epoch": 1.903576096792573,
+ "grad_norm": 0.21045279502868652,
+ "learning_rate": 7.258713136729223e-07,
+ "loss": 0.4943391799926758,
+ "step": 7100
+ },
+ {
+ "epoch": 1.9089385662097396,
+ "grad_norm": 0.18638195097446442,
+ "learning_rate": 6.85656836461126e-07,
+ "loss": 0.48406500816345216,
+ "step": 7120
+ },
+ {
+ "epoch": 1.9143010356269063,
+ "grad_norm": 0.15428832173347473,
+ "learning_rate": 6.454423592493298e-07,
+ "loss": 0.5084923267364502,
+ "step": 7140
+ },
+ {
+ "epoch": 1.9196635050440727,
+ "grad_norm": 0.2415294051170349,
+ "learning_rate": 6.052278820375336e-07,
+ "loss": 0.5026498317718506,
+ "step": 7160
+ },
+ {
+ "epoch": 1.9250259744612395,
+ "grad_norm": 0.23021087050437927,
+ "learning_rate": 5.650134048257373e-07,
+ "loss": 0.5294596195220947,
+ "step": 7180
+ },
+ {
+ "epoch": 1.930388443878406,
+ "grad_norm": 0.21689893305301666,
+ "learning_rate": 5.24798927613941e-07,
+ "loss": 0.4569683074951172,
+ "step": 7200
+ },
+ {
+ "epoch": 1.9357509132955726,
+ "grad_norm": 0.25458022952079773,
+ "learning_rate": 4.845844504021448e-07,
+ "loss": 0.4905412197113037,
+ "step": 7220
+ },
+ {
+ "epoch": 1.9411133827127394,
+ "grad_norm": 0.20990079641342163,
+ "learning_rate": 4.4436997319034854e-07,
+ "loss": 0.49599390029907225,
+ "step": 7240
+ },
+ {
+ "epoch": 1.9464758521299057,
+ "grad_norm": 0.2381497025489807,
+ "learning_rate": 4.041554959785523e-07,
+ "loss": 0.5149903774261475,
+ "step": 7260
+ },
+ {
+ "epoch": 1.9518383215470725,
+ "grad_norm": 0.29006749391555786,
+ "learning_rate": 3.6394101876675604e-07,
+ "loss": 0.5063531875610352,
+ "step": 7280
+ },
+ {
+ "epoch": 1.957200790964239,
+ "grad_norm": 0.20159471035003662,
+ "learning_rate": 3.237265415549598e-07,
+ "loss": 0.5062472343444824,
+ "step": 7300
+ },
+ {
+ "epoch": 1.9625632603814056,
+ "grad_norm": 0.21182367205619812,
+ "learning_rate": 2.8351206434316354e-07,
+ "loss": 0.49964118003845215,
+ "step": 7320
+ },
+ {
+ "epoch": 1.9679257297985724,
+ "grad_norm": 0.28606927394866943,
+ "learning_rate": 2.432975871313673e-07,
+ "loss": 0.48647170066833495,
+ "step": 7340
+ },
+ {
+ "epoch": 1.9732881992157387,
+ "grad_norm": 0.2779375910758972,
+ "learning_rate": 2.0308310991957104e-07,
+ "loss": 0.4816920280456543,
+ "step": 7360
+ },
+ {
+ "epoch": 1.9786506686329055,
+ "grad_norm": 0.20412451028823853,
+ "learning_rate": 1.628686327077748e-07,
+ "loss": 0.5177248477935791,
+ "step": 7380
+ },
+ {
+ "epoch": 1.984013138050072,
+ "grad_norm": 0.19861914217472076,
+ "learning_rate": 1.2265415549597854e-07,
+ "loss": 0.5265318870544433,
+ "step": 7400
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 9.118750722760274e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-7400/training_args.bin b/checkpoint-7400/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-7400/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-7460/README.md b/checkpoint-7460/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-7460/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-7460/adapter_config.json b/checkpoint-7460/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-7460/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-7460/adapter_model.safetensors b/checkpoint-7460/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..2d56dd52e04576c533bceb18ec394b50943e9d91
--- /dev/null
+++ b/checkpoint-7460/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e45b98549709b9485f32de70adf0f96edce8ca97b599382c87a97973290a9c87
+size 50360752
diff --git a/checkpoint-7460/chat_template.jinja b/checkpoint-7460/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-7460/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-7460/optimizer.pt b/checkpoint-7460/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..1a2eeab20f259c7ed54f824a514fd2254a47fbc1
--- /dev/null
+++ b/checkpoint-7460/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e4e64fa3c707df178cda23d639abb1f1cafc21e2dbf509d522cfd8f580758a69
+size 100828235
diff --git a/checkpoint-7460/rng_state.pth b/checkpoint-7460/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..38d00ee5d06e3f13ef1759ccbdb3543090dd7260
--- /dev/null
+++ b/checkpoint-7460/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:823c0df99ec859ef099d61782766170fdc56bce9968f4ae05385de24a019d319
+size 14645
diff --git a/checkpoint-7460/scheduler.pt b/checkpoint-7460/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..d30cf94abd4985514130e141607ae9ab5086ea1e
--- /dev/null
+++ b/checkpoint-7460/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0e08f38c509bf4e9ef67661f9a81b33e13d2563c8cb40e2dfda7539d931c734d
+size 1465
diff --git a/checkpoint-7460/tokenizer.json b/checkpoint-7460/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-7460/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-7460/tokenizer_config.json b/checkpoint-7460/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-7460/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-7460/trainer_state.json b/checkpoint-7460/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..078db794fc37dbb4d6ec0624d30d01b51bab32e5
--- /dev/null
+++ b/checkpoint-7460/trainer_state.json
@@ -0,0 +1,2645 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 2.0,
+ "eval_steps": 500,
+ "global_step": 7460,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ },
+ {
+ "epoch": 0.21986124610383082,
+ "grad_norm": 0.08678701519966125,
+ "learning_rate": 1.3353217158176944e-05,
+ "loss": 0.5999309062957764,
+ "step": 820
+ },
+ {
+ "epoch": 0.22522371552099743,
+ "grad_norm": 0.12222636491060257,
+ "learning_rate": 1.3313002680965148e-05,
+ "loss": 0.5421838760375977,
+ "step": 840
+ },
+ {
+ "epoch": 0.23058618493816402,
+ "grad_norm": 0.11634483933448792,
+ "learning_rate": 1.3272788203753352e-05,
+ "loss": 0.6069926261901856,
+ "step": 860
+ },
+ {
+ "epoch": 0.23594865435533063,
+ "grad_norm": 0.12163955718278885,
+ "learning_rate": 1.3232573726541556e-05,
+ "loss": 0.5558357238769531,
+ "step": 880
+ },
+ {
+ "epoch": 0.24131112377249722,
+ "grad_norm": 0.13140572607517242,
+ "learning_rate": 1.319235924932976e-05,
+ "loss": 0.5537341117858887,
+ "step": 900
+ },
+ {
+ "epoch": 0.24667359318966384,
+ "grad_norm": 0.1295424848794937,
+ "learning_rate": 1.3152144772117963e-05,
+ "loss": 0.5734247684478759,
+ "step": 920
+ },
+ {
+ "epoch": 0.2520360626068304,
+ "grad_norm": 0.08855397999286652,
+ "learning_rate": 1.3111930294906167e-05,
+ "loss": 0.5499854564666748,
+ "step": 940
+ },
+ {
+ "epoch": 0.25739853202399704,
+ "grad_norm": 0.10895389318466187,
+ "learning_rate": 1.307171581769437e-05,
+ "loss": 0.4994966506958008,
+ "step": 960
+ },
+ {
+ "epoch": 0.26276100144116366,
+ "grad_norm": 0.10110122710466385,
+ "learning_rate": 1.3031501340482574e-05,
+ "loss": 0.5803254604339599,
+ "step": 980
+ },
+ {
+ "epoch": 0.26812347085833027,
+ "grad_norm": 0.1323656141757965,
+ "learning_rate": 1.2991286863270778e-05,
+ "loss": 0.5268758773803711,
+ "step": 1000
+ },
+ {
+ "epoch": 0.2734859402754969,
+ "grad_norm": 0.09068968147039413,
+ "learning_rate": 1.2951072386058981e-05,
+ "loss": 0.5150487899780274,
+ "step": 1020
+ },
+ {
+ "epoch": 0.27884840969266345,
+ "grad_norm": 0.11400057375431061,
+ "learning_rate": 1.2910857908847185e-05,
+ "loss": 0.5365507125854492,
+ "step": 1040
+ },
+ {
+ "epoch": 0.28421087910983006,
+ "grad_norm": 0.14133770763874054,
+ "learning_rate": 1.2870643431635389e-05,
+ "loss": 0.5134270668029786,
+ "step": 1060
+ },
+ {
+ "epoch": 0.2895733485269967,
+ "grad_norm": 0.14621631801128387,
+ "learning_rate": 1.2830428954423593e-05,
+ "loss": 0.5870331287384033,
+ "step": 1080
+ },
+ {
+ "epoch": 0.2949358179441633,
+ "grad_norm": 0.09397239238023758,
+ "learning_rate": 1.2790214477211796e-05,
+ "loss": 0.5265964984893798,
+ "step": 1100
+ },
+ {
+ "epoch": 0.3002982873613299,
+ "grad_norm": 0.13457220792770386,
+ "learning_rate": 1.275e-05,
+ "loss": 0.541674280166626,
+ "step": 1120
+ },
+ {
+ "epoch": 0.3056607567784965,
+ "grad_norm": 0.11553078144788742,
+ "learning_rate": 1.2709785522788204e-05,
+ "loss": 0.5721035003662109,
+ "step": 1140
+ },
+ {
+ "epoch": 0.3110232261956631,
+ "grad_norm": 0.08464279770851135,
+ "learning_rate": 1.2669571045576407e-05,
+ "loss": 0.5242496967315674,
+ "step": 1160
+ },
+ {
+ "epoch": 0.3163856956128297,
+ "grad_norm": 0.11578533798456192,
+ "learning_rate": 1.2629356568364611e-05,
+ "loss": 0.5268265724182128,
+ "step": 1180
+ },
+ {
+ "epoch": 0.3217481650299963,
+ "grad_norm": 0.10422660410404205,
+ "learning_rate": 1.2589142091152815e-05,
+ "loss": 0.5755553722381592,
+ "step": 1200
+ },
+ {
+ "epoch": 0.32711063444716293,
+ "grad_norm": 0.1601565182209015,
+ "learning_rate": 1.2548927613941018e-05,
+ "loss": 0.572784423828125,
+ "step": 1220
+ },
+ {
+ "epoch": 0.33247310386432954,
+ "grad_norm": 0.1435895711183548,
+ "learning_rate": 1.2508713136729222e-05,
+ "loss": 0.4759331703186035,
+ "step": 1240
+ },
+ {
+ "epoch": 0.3378355732814961,
+ "grad_norm": 0.13164320588111877,
+ "learning_rate": 1.2468498659517426e-05,
+ "loss": 0.5674447059631348,
+ "step": 1260
+ },
+ {
+ "epoch": 0.3431980426986627,
+ "grad_norm": 0.17907585203647614,
+ "learning_rate": 1.242828418230563e-05,
+ "loss": 0.5384601593017578,
+ "step": 1280
+ },
+ {
+ "epoch": 0.34856051211582934,
+ "grad_norm": 0.1515372097492218,
+ "learning_rate": 1.2388069705093833e-05,
+ "loss": 0.5154921531677246,
+ "step": 1300
+ },
+ {
+ "epoch": 0.35392298153299595,
+ "grad_norm": 0.13605119287967682,
+ "learning_rate": 1.2347855227882037e-05,
+ "loss": 0.5586633205413818,
+ "step": 1320
+ },
+ {
+ "epoch": 0.35928545095016257,
+ "grad_norm": 0.12003476917743683,
+ "learning_rate": 1.230764075067024e-05,
+ "loss": 0.5512509822845459,
+ "step": 1340
+ },
+ {
+ "epoch": 0.3646479203673292,
+ "grad_norm": 0.11852169036865234,
+ "learning_rate": 1.2267426273458444e-05,
+ "loss": 0.5680348873138428,
+ "step": 1360
+ },
+ {
+ "epoch": 0.37001038978449574,
+ "grad_norm": 0.16344694793224335,
+ "learning_rate": 1.2227211796246648e-05,
+ "loss": 0.5669443130493164,
+ "step": 1380
+ },
+ {
+ "epoch": 0.37537285920166236,
+ "grad_norm": 0.11730384081602097,
+ "learning_rate": 1.2186997319034852e-05,
+ "loss": 0.5089732646942139,
+ "step": 1400
+ },
+ {
+ "epoch": 0.38073532861882897,
+ "grad_norm": 0.1063583567738533,
+ "learning_rate": 1.2146782841823055e-05,
+ "loss": 0.5337563037872315,
+ "step": 1420
+ },
+ {
+ "epoch": 0.3860977980359956,
+ "grad_norm": 0.12790119647979736,
+ "learning_rate": 1.2106568364611259e-05,
+ "loss": 0.5077777862548828,
+ "step": 1440
+ },
+ {
+ "epoch": 0.3914602674531622,
+ "grad_norm": 0.1386743038892746,
+ "learning_rate": 1.2066353887399463e-05,
+ "loss": 0.5521824836730957,
+ "step": 1460
+ },
+ {
+ "epoch": 0.39682273687032876,
+ "grad_norm": 0.0992259532213211,
+ "learning_rate": 1.2026139410187666e-05,
+ "loss": 0.554673147201538,
+ "step": 1480
+ },
+ {
+ "epoch": 0.4021852062874954,
+ "grad_norm": 0.15981841087341309,
+ "learning_rate": 1.1985924932975872e-05,
+ "loss": 0.5779122352600098,
+ "step": 1500
+ },
+ {
+ "epoch": 0.407547675704662,
+ "grad_norm": 0.19671906530857086,
+ "learning_rate": 1.1945710455764076e-05,
+ "loss": 0.5743378162384033,
+ "step": 1520
+ },
+ {
+ "epoch": 0.4129101451218286,
+ "grad_norm": 0.10725795477628708,
+ "learning_rate": 1.190549597855228e-05,
+ "loss": 0.523157787322998,
+ "step": 1540
+ },
+ {
+ "epoch": 0.4182726145389952,
+ "grad_norm": 0.14457851648330688,
+ "learning_rate": 1.1865281501340483e-05,
+ "loss": 0.5441864490509033,
+ "step": 1560
+ },
+ {
+ "epoch": 0.42363508395616184,
+ "grad_norm": 0.15479697287082672,
+ "learning_rate": 1.1825067024128687e-05,
+ "loss": 0.6409400463104248,
+ "step": 1580
+ },
+ {
+ "epoch": 0.4289975533733284,
+ "grad_norm": 0.11132492870092392,
+ "learning_rate": 1.178485254691689e-05,
+ "loss": 0.5462933540344238,
+ "step": 1600
+ },
+ {
+ "epoch": 0.434360022790495,
+ "grad_norm": 0.11062806099653244,
+ "learning_rate": 1.1744638069705094e-05,
+ "loss": 0.5428354740142822,
+ "step": 1620
+ },
+ {
+ "epoch": 0.43972249220766163,
+ "grad_norm": 0.1327652931213379,
+ "learning_rate": 1.1704423592493298e-05,
+ "loss": 0.5324414253234864,
+ "step": 1640
+ },
+ {
+ "epoch": 0.44508496162482825,
+ "grad_norm": 0.1209583580493927,
+ "learning_rate": 1.1664209115281501e-05,
+ "loss": 0.5270706176757812,
+ "step": 1660
+ },
+ {
+ "epoch": 0.45044743104199486,
+ "grad_norm": 0.11154980212450027,
+ "learning_rate": 1.1623994638069705e-05,
+ "loss": 0.525149154663086,
+ "step": 1680
+ },
+ {
+ "epoch": 0.4558099004591614,
+ "grad_norm": 0.14099697768688202,
+ "learning_rate": 1.158378016085791e-05,
+ "loss": 0.5981990814208984,
+ "step": 1700
+ },
+ {
+ "epoch": 0.46117236987632804,
+ "grad_norm": 0.11787982285022736,
+ "learning_rate": 1.1543565683646114e-05,
+ "loss": 0.5327546119689941,
+ "step": 1720
+ },
+ {
+ "epoch": 0.46653483929349465,
+ "grad_norm": 0.12584130465984344,
+ "learning_rate": 1.1503351206434318e-05,
+ "loss": 0.5126790046691895,
+ "step": 1740
+ },
+ {
+ "epoch": 0.47189730871066127,
+ "grad_norm": 0.16248232126235962,
+ "learning_rate": 1.1463136729222522e-05,
+ "loss": 0.5697287082672119,
+ "step": 1760
+ },
+ {
+ "epoch": 0.4772597781278279,
+ "grad_norm": 0.14940819144248962,
+ "learning_rate": 1.1422922252010725e-05,
+ "loss": 0.5015492916107178,
+ "step": 1780
+ },
+ {
+ "epoch": 0.48262224754499444,
+ "grad_norm": 0.1647220402956009,
+ "learning_rate": 1.1382707774798929e-05,
+ "loss": 0.5097331523895263,
+ "step": 1800
+ },
+ {
+ "epoch": 0.48798471696216106,
+ "grad_norm": 0.12255030870437622,
+ "learning_rate": 1.1342493297587133e-05,
+ "loss": 0.5670981407165527,
+ "step": 1820
+ },
+ {
+ "epoch": 0.4933471863793277,
+ "grad_norm": 0.1160770058631897,
+ "learning_rate": 1.1302278820375336e-05,
+ "loss": 0.5236512660980225,
+ "step": 1840
+ },
+ {
+ "epoch": 0.4987096557964943,
+ "grad_norm": 0.21711941063404083,
+ "learning_rate": 1.126206434316354e-05,
+ "loss": 0.5926671504974366,
+ "step": 1860
+ },
+ {
+ "epoch": 0.5040721252136608,
+ "grad_norm": 0.16682052612304688,
+ "learning_rate": 1.1221849865951744e-05,
+ "loss": 0.5240281581878662,
+ "step": 1880
+ },
+ {
+ "epoch": 0.5094345946308275,
+ "grad_norm": 0.16348475217819214,
+ "learning_rate": 1.1181635388739948e-05,
+ "loss": 0.5574026107788086,
+ "step": 1900
+ },
+ {
+ "epoch": 0.5147970640479941,
+ "grad_norm": 0.17506958544254303,
+ "learning_rate": 1.1141420911528151e-05,
+ "loss": 0.5592098236083984,
+ "step": 1920
+ },
+ {
+ "epoch": 0.5201595334651608,
+ "grad_norm": 0.1784403771162033,
+ "learning_rate": 1.1101206434316355e-05,
+ "loss": 0.5189618110656739,
+ "step": 1940
+ },
+ {
+ "epoch": 0.5255220028823273,
+ "grad_norm": 0.17252163589000702,
+ "learning_rate": 1.1060991957104559e-05,
+ "loss": 0.5126346111297607,
+ "step": 1960
+ },
+ {
+ "epoch": 0.5308844722994939,
+ "grad_norm": 0.12690365314483643,
+ "learning_rate": 1.1020777479892762e-05,
+ "loss": 0.5473652362823487,
+ "step": 1980
+ },
+ {
+ "epoch": 0.5362469417166605,
+ "grad_norm": 0.1284744292497635,
+ "learning_rate": 1.0980563002680966e-05,
+ "loss": 0.5309309482574462,
+ "step": 2000
+ },
+ {
+ "epoch": 0.5416094111338271,
+ "grad_norm": 0.1850503385066986,
+ "learning_rate": 1.094034852546917e-05,
+ "loss": 0.5636833190917969,
+ "step": 2020
+ },
+ {
+ "epoch": 0.5469718805509938,
+ "grad_norm": 0.1514296680688858,
+ "learning_rate": 1.0900134048257373e-05,
+ "loss": 0.5273778915405274,
+ "step": 2040
+ },
+ {
+ "epoch": 0.5523343499681603,
+ "grad_norm": 0.1502915471792221,
+ "learning_rate": 1.0859919571045577e-05,
+ "loss": 0.6000364780426025,
+ "step": 2060
+ },
+ {
+ "epoch": 0.5576968193853269,
+ "grad_norm": 0.14147423207759857,
+ "learning_rate": 1.081970509383378e-05,
+ "loss": 0.5480428218841553,
+ "step": 2080
+ },
+ {
+ "epoch": 0.5630592888024936,
+ "grad_norm": 0.13399621844291687,
+ "learning_rate": 1.0779490616621984e-05,
+ "loss": 0.513938045501709,
+ "step": 2100
+ },
+ {
+ "epoch": 0.5684217582196601,
+ "grad_norm": 0.12856991589069366,
+ "learning_rate": 1.0739276139410188e-05,
+ "loss": 0.4760735988616943,
+ "step": 2120
+ },
+ {
+ "epoch": 0.5737842276368268,
+ "grad_norm": 0.15576769411563873,
+ "learning_rate": 1.0699061662198392e-05,
+ "loss": 0.5474783420562744,
+ "step": 2140
+ },
+ {
+ "epoch": 0.5791466970539934,
+ "grad_norm": 0.2024153470993042,
+ "learning_rate": 1.0658847184986596e-05,
+ "loss": 0.5309592723846436,
+ "step": 2160
+ },
+ {
+ "epoch": 0.58450916647116,
+ "grad_norm": 0.13033868372440338,
+ "learning_rate": 1.06186327077748e-05,
+ "loss": 0.5345770835876464,
+ "step": 2180
+ },
+ {
+ "epoch": 0.5898716358883266,
+ "grad_norm": 0.15354423224925995,
+ "learning_rate": 1.0578418230563003e-05,
+ "loss": 0.5441046714782715,
+ "step": 2200
+ },
+ {
+ "epoch": 0.5952341053054931,
+ "grad_norm": 0.19533827900886536,
+ "learning_rate": 1.0538203753351207e-05,
+ "loss": 0.547668170928955,
+ "step": 2220
+ },
+ {
+ "epoch": 0.6005965747226598,
+ "grad_norm": 0.15901635587215424,
+ "learning_rate": 1.049798927613941e-05,
+ "loss": 0.5213536739349365,
+ "step": 2240
+ },
+ {
+ "epoch": 0.6059590441398264,
+ "grad_norm": 0.20392107963562012,
+ "learning_rate": 1.0457774798927614e-05,
+ "loss": 0.56328444480896,
+ "step": 2260
+ },
+ {
+ "epoch": 0.611321513556993,
+ "grad_norm": 0.14985501766204834,
+ "learning_rate": 1.0417560321715818e-05,
+ "loss": 0.5592964172363282,
+ "step": 2280
+ },
+ {
+ "epoch": 0.6166839829741596,
+ "grad_norm": 0.16292506456375122,
+ "learning_rate": 1.0377345844504021e-05,
+ "loss": 0.6026081562042236,
+ "step": 2300
+ },
+ {
+ "epoch": 0.6220464523913262,
+ "grad_norm": 0.2114475965499878,
+ "learning_rate": 1.0337131367292225e-05,
+ "loss": 0.5434895992279053,
+ "step": 2320
+ },
+ {
+ "epoch": 0.6274089218084928,
+ "grad_norm": 0.15036092698574066,
+ "learning_rate": 1.0296916890080429e-05,
+ "loss": 0.5241796016693115,
+ "step": 2340
+ },
+ {
+ "epoch": 0.6327713912256594,
+ "grad_norm": 0.2040790617465973,
+ "learning_rate": 1.0256702412868633e-05,
+ "loss": 0.5172519683837891,
+ "step": 2360
+ },
+ {
+ "epoch": 0.6381338606428261,
+ "grad_norm": 0.15708747506141663,
+ "learning_rate": 1.0216487935656836e-05,
+ "loss": 0.49505252838134767,
+ "step": 2380
+ },
+ {
+ "epoch": 0.6434963300599926,
+ "grad_norm": 0.1831217259168625,
+ "learning_rate": 1.017627345844504e-05,
+ "loss": 0.5166856288909912,
+ "step": 2400
+ },
+ {
+ "epoch": 0.6488587994771592,
+ "grad_norm": 0.23026946187019348,
+ "learning_rate": 1.0136058981233244e-05,
+ "loss": 0.5275045394897461,
+ "step": 2420
+ },
+ {
+ "epoch": 0.6542212688943259,
+ "grad_norm": 0.17848673462867737,
+ "learning_rate": 1.0095844504021447e-05,
+ "loss": 0.5764461994171143,
+ "step": 2440
+ },
+ {
+ "epoch": 0.6595837383114924,
+ "grad_norm": 0.14768671989440918,
+ "learning_rate": 1.0055630026809651e-05,
+ "loss": 0.4772446632385254,
+ "step": 2460
+ },
+ {
+ "epoch": 0.6649462077286591,
+ "grad_norm": 0.11061226576566696,
+ "learning_rate": 1.0015415549597855e-05,
+ "loss": 0.4822176456451416,
+ "step": 2480
+ },
+ {
+ "epoch": 0.6703086771458256,
+ "grad_norm": 0.22382384538650513,
+ "learning_rate": 9.975201072386058e-06,
+ "loss": 0.5523125648498535,
+ "step": 2500
+ },
+ {
+ "epoch": 0.6756711465629922,
+ "grad_norm": 0.1481855809688568,
+ "learning_rate": 9.934986595174262e-06,
+ "loss": 0.5522858619689941,
+ "step": 2520
+ },
+ {
+ "epoch": 0.6810336159801589,
+ "grad_norm": 0.16584496200084686,
+ "learning_rate": 9.894772117962466e-06,
+ "loss": 0.5220115661621094,
+ "step": 2540
+ },
+ {
+ "epoch": 0.6863960853973254,
+ "grad_norm": 0.24747292697429657,
+ "learning_rate": 9.85455764075067e-06,
+ "loss": 0.5106014728546142,
+ "step": 2560
+ },
+ {
+ "epoch": 0.6917585548144921,
+ "grad_norm": 0.1886838674545288,
+ "learning_rate": 9.814343163538873e-06,
+ "loss": 0.554722261428833,
+ "step": 2580
+ },
+ {
+ "epoch": 0.6971210242316587,
+ "grad_norm": 0.14403431117534637,
+ "learning_rate": 9.774128686327077e-06,
+ "loss": 0.5226208209991455,
+ "step": 2600
+ },
+ {
+ "epoch": 0.7024834936488252,
+ "grad_norm": 0.1577453911304474,
+ "learning_rate": 9.73391420911528e-06,
+ "loss": 0.5295976161956787,
+ "step": 2620
+ },
+ {
+ "epoch": 0.7078459630659919,
+ "grad_norm": 0.2269749790430069,
+ "learning_rate": 9.693699731903484e-06,
+ "loss": 0.5336898803710938,
+ "step": 2640
+ },
+ {
+ "epoch": 0.7132084324831585,
+ "grad_norm": 0.23890693485736847,
+ "learning_rate": 9.653485254691688e-06,
+ "loss": 0.5564133644104003,
+ "step": 2660
+ },
+ {
+ "epoch": 0.7185709019003251,
+ "grad_norm": 0.19051003456115723,
+ "learning_rate": 9.613270777479892e-06,
+ "loss": 0.5483838081359863,
+ "step": 2680
+ },
+ {
+ "epoch": 0.7239333713174917,
+ "grad_norm": 0.15244685113430023,
+ "learning_rate": 9.573056300268095e-06,
+ "loss": 0.5657371520996094,
+ "step": 2700
+ },
+ {
+ "epoch": 0.7292958407346584,
+ "grad_norm": 0.14131584763526917,
+ "learning_rate": 9.532841823056299e-06,
+ "loss": 0.5375633716583252,
+ "step": 2720
+ },
+ {
+ "epoch": 0.7346583101518249,
+ "grad_norm": 0.15706594288349152,
+ "learning_rate": 9.492627345844505e-06,
+ "loss": 0.5774847507476807,
+ "step": 2740
+ },
+ {
+ "epoch": 0.7400207795689915,
+ "grad_norm": 0.120318703353405,
+ "learning_rate": 9.452412868632708e-06,
+ "loss": 0.5289290428161622,
+ "step": 2760
+ },
+ {
+ "epoch": 0.7453832489861582,
+ "grad_norm": 0.17643575370311737,
+ "learning_rate": 9.412198391420912e-06,
+ "loss": 0.548846435546875,
+ "step": 2780
+ },
+ {
+ "epoch": 0.7507457184033247,
+ "grad_norm": 0.23063655197620392,
+ "learning_rate": 9.371983914209116e-06,
+ "loss": 0.5502467155456543,
+ "step": 2800
+ },
+ {
+ "epoch": 0.7561081878204914,
+ "grad_norm": 0.14489713311195374,
+ "learning_rate": 9.33176943699732e-06,
+ "loss": 0.5205071449279786,
+ "step": 2820
+ },
+ {
+ "epoch": 0.7614706572376579,
+ "grad_norm": 0.15738680958747864,
+ "learning_rate": 9.291554959785523e-06,
+ "loss": 0.5463311195373535,
+ "step": 2840
+ },
+ {
+ "epoch": 0.7668331266548245,
+ "grad_norm": 0.1291189193725586,
+ "learning_rate": 9.251340482573727e-06,
+ "loss": 0.5183065414428711,
+ "step": 2860
+ },
+ {
+ "epoch": 0.7721955960719912,
+ "grad_norm": 0.14537270367145538,
+ "learning_rate": 9.21112600536193e-06,
+ "loss": 0.5544816493988037,
+ "step": 2880
+ },
+ {
+ "epoch": 0.7775580654891577,
+ "grad_norm": 0.13409097492694855,
+ "learning_rate": 9.170911528150134e-06,
+ "loss": 0.5107351303100586,
+ "step": 2900
+ },
+ {
+ "epoch": 0.7829205349063244,
+ "grad_norm": 0.2998020052909851,
+ "learning_rate": 9.130697050938338e-06,
+ "loss": 0.5310684680938721,
+ "step": 2920
+ },
+ {
+ "epoch": 0.788283004323491,
+ "grad_norm": 0.1838223934173584,
+ "learning_rate": 9.090482573726543e-06,
+ "loss": 0.5270499229431153,
+ "step": 2940
+ },
+ {
+ "epoch": 0.7936454737406575,
+ "grad_norm": 0.18618327379226685,
+ "learning_rate": 9.050268096514747e-06,
+ "loss": 0.5336289882659913,
+ "step": 2960
+ },
+ {
+ "epoch": 0.7990079431578242,
+ "grad_norm": 0.20681297779083252,
+ "learning_rate": 9.01005361930295e-06,
+ "loss": 0.508507251739502,
+ "step": 2980
+ },
+ {
+ "epoch": 0.8043704125749908,
+ "grad_norm": 0.24283935129642487,
+ "learning_rate": 8.969839142091154e-06,
+ "loss": 0.5339189052581788,
+ "step": 3000
+ },
+ {
+ "epoch": 0.8097328819921574,
+ "grad_norm": 0.21722275018692017,
+ "learning_rate": 8.929624664879358e-06,
+ "loss": 0.515669584274292,
+ "step": 3020
+ },
+ {
+ "epoch": 0.815095351409324,
+ "grad_norm": 0.14678969979286194,
+ "learning_rate": 8.889410187667562e-06,
+ "loss": 0.49359521865844724,
+ "step": 3040
+ },
+ {
+ "epoch": 0.8204578208264905,
+ "grad_norm": 0.16017946600914001,
+ "learning_rate": 8.849195710455765e-06,
+ "loss": 0.532757043838501,
+ "step": 3060
+ },
+ {
+ "epoch": 0.8258202902436572,
+ "grad_norm": 0.13103698194026947,
+ "learning_rate": 8.808981233243969e-06,
+ "loss": 0.5174227237701416,
+ "step": 3080
+ },
+ {
+ "epoch": 0.8311827596608238,
+ "grad_norm": 0.13764740526676178,
+ "learning_rate": 8.768766756032173e-06,
+ "loss": 0.5756002902984619,
+ "step": 3100
+ },
+ {
+ "epoch": 0.8365452290779904,
+ "grad_norm": 0.1956685334444046,
+ "learning_rate": 8.728552278820376e-06,
+ "loss": 0.5458150386810303,
+ "step": 3120
+ },
+ {
+ "epoch": 0.841907698495157,
+ "grad_norm": 0.14859093725681305,
+ "learning_rate": 8.68833780160858e-06,
+ "loss": 0.5232916831970215,
+ "step": 3140
+ },
+ {
+ "epoch": 0.8472701679123237,
+ "grad_norm": 0.14078572392463684,
+ "learning_rate": 8.648123324396784e-06,
+ "loss": 0.45665884017944336,
+ "step": 3160
+ },
+ {
+ "epoch": 0.8526326373294902,
+ "grad_norm": 0.10593896359205246,
+ "learning_rate": 8.607908847184988e-06,
+ "loss": 0.46901817321777345,
+ "step": 3180
+ },
+ {
+ "epoch": 0.8579951067466568,
+ "grad_norm": 0.19927014410495758,
+ "learning_rate": 8.567694369973191e-06,
+ "loss": 0.4962503910064697,
+ "step": 3200
+ },
+ {
+ "epoch": 0.8633575761638235,
+ "grad_norm": 0.1885233223438263,
+ "learning_rate": 8.527479892761395e-06,
+ "loss": 0.5428553581237793,
+ "step": 3220
+ },
+ {
+ "epoch": 0.86872004558099,
+ "grad_norm": 0.22774286568164825,
+ "learning_rate": 8.487265415549599e-06,
+ "loss": 0.5246198177337646,
+ "step": 3240
+ },
+ {
+ "epoch": 0.8740825149981567,
+ "grad_norm": 0.16228961944580078,
+ "learning_rate": 8.447050938337802e-06,
+ "loss": 0.5317719936370849,
+ "step": 3260
+ },
+ {
+ "epoch": 0.8794449844153233,
+ "grad_norm": 0.19011476635932922,
+ "learning_rate": 8.406836461126006e-06,
+ "loss": 0.5377527236938476,
+ "step": 3280
+ },
+ {
+ "epoch": 0.8848074538324898,
+ "grad_norm": 0.1937844604253769,
+ "learning_rate": 8.36662198391421e-06,
+ "loss": 0.5009727954864502,
+ "step": 3300
+ },
+ {
+ "epoch": 0.8901699232496565,
+ "grad_norm": 0.26362502574920654,
+ "learning_rate": 8.326407506702413e-06,
+ "loss": 0.5286832809448242,
+ "step": 3320
+ },
+ {
+ "epoch": 0.895532392666823,
+ "grad_norm": 0.15528951585292816,
+ "learning_rate": 8.286193029490617e-06,
+ "loss": 0.5699362754821777,
+ "step": 3340
+ },
+ {
+ "epoch": 0.9008948620839897,
+ "grad_norm": 0.19824309647083282,
+ "learning_rate": 8.24597855227882e-06,
+ "loss": 0.5417330265045166,
+ "step": 3360
+ },
+ {
+ "epoch": 0.9062573315011563,
+ "grad_norm": 0.17824552953243256,
+ "learning_rate": 8.205764075067025e-06,
+ "loss": 0.5166538238525391,
+ "step": 3380
+ },
+ {
+ "epoch": 0.9116198009183228,
+ "grad_norm": 0.1860542744398117,
+ "learning_rate": 8.165549597855228e-06,
+ "loss": 0.5525233745574951,
+ "step": 3400
+ },
+ {
+ "epoch": 0.9169822703354895,
+ "grad_norm": 0.22200629115104675,
+ "learning_rate": 8.125335120643432e-06,
+ "loss": 0.48862462043762206,
+ "step": 3420
+ },
+ {
+ "epoch": 0.9223447397526561,
+ "grad_norm": 0.21177783608436584,
+ "learning_rate": 8.085120643431636e-06,
+ "loss": 0.5362657070159912,
+ "step": 3440
+ },
+ {
+ "epoch": 0.9277072091698227,
+ "grad_norm": 0.1278514564037323,
+ "learning_rate": 8.04490616621984e-06,
+ "loss": 0.5472875595092773,
+ "step": 3460
+ },
+ {
+ "epoch": 0.9330696785869893,
+ "grad_norm": 0.1520422250032425,
+ "learning_rate": 8.004691689008043e-06,
+ "loss": 0.4906148910522461,
+ "step": 3480
+ },
+ {
+ "epoch": 0.9384321480041559,
+ "grad_norm": 0.1678784340620041,
+ "learning_rate": 7.964477211796247e-06,
+ "loss": 0.5190341949462891,
+ "step": 3500
+ },
+ {
+ "epoch": 0.9437946174213225,
+ "grad_norm": 0.2168162763118744,
+ "learning_rate": 7.92426273458445e-06,
+ "loss": 0.5007696151733398,
+ "step": 3520
+ },
+ {
+ "epoch": 0.9491570868384891,
+ "grad_norm": 0.18424147367477417,
+ "learning_rate": 7.884048257372654e-06,
+ "loss": 0.5395221710205078,
+ "step": 3540
+ },
+ {
+ "epoch": 0.9545195562556558,
+ "grad_norm": 0.17553555965423584,
+ "learning_rate": 7.843833780160858e-06,
+ "loss": 0.4716806888580322,
+ "step": 3560
+ },
+ {
+ "epoch": 0.9598820256728223,
+ "grad_norm": 0.15070843696594238,
+ "learning_rate": 7.803619302949062e-06,
+ "loss": 0.49967169761657715,
+ "step": 3580
+ },
+ {
+ "epoch": 0.9652444950899889,
+ "grad_norm": 0.172193244099617,
+ "learning_rate": 7.763404825737265e-06,
+ "loss": 0.495190954208374,
+ "step": 3600
+ },
+ {
+ "epoch": 0.9706069645071556,
+ "grad_norm": 0.15822157263755798,
+ "learning_rate": 7.723190348525469e-06,
+ "loss": 0.5322632789611816,
+ "step": 3620
+ },
+ {
+ "epoch": 0.9759694339243221,
+ "grad_norm": 0.19345910847187042,
+ "learning_rate": 7.682975871313673e-06,
+ "loss": 0.48404436111450194,
+ "step": 3640
+ },
+ {
+ "epoch": 0.9813319033414888,
+ "grad_norm": 0.17885969579219818,
+ "learning_rate": 7.642761394101876e-06,
+ "loss": 0.5166211128234863,
+ "step": 3660
+ },
+ {
+ "epoch": 0.9866943727586553,
+ "grad_norm": 0.15497833490371704,
+ "learning_rate": 7.60254691689008e-06,
+ "loss": 0.5560059547424316,
+ "step": 3680
+ },
+ {
+ "epoch": 0.992056842175822,
+ "grad_norm": 0.17155644297599792,
+ "learning_rate": 7.562332439678284e-06,
+ "loss": 0.529679822921753,
+ "step": 3700
+ },
+ {
+ "epoch": 0.9974193115929886,
+ "grad_norm": 0.18267494440078735,
+ "learning_rate": 7.522117962466487e-06,
+ "loss": 0.5055463790893555,
+ "step": 3720
+ },
+ {
+ "epoch": 1.0026812347085834,
+ "grad_norm": 0.1627507209777832,
+ "learning_rate": 7.481903485254692e-06,
+ "loss": 0.45867152214050294,
+ "step": 3740
+ },
+ {
+ "epoch": 1.00804370412575,
+ "grad_norm": 0.2230822890996933,
+ "learning_rate": 7.441689008042896e-06,
+ "loss": 0.4909696102142334,
+ "step": 3760
+ },
+ {
+ "epoch": 1.0134061735429165,
+ "grad_norm": 0.14418569207191467,
+ "learning_rate": 7.401474530831099e-06,
+ "loss": 0.4891301155090332,
+ "step": 3780
+ },
+ {
+ "epoch": 1.018768642960083,
+ "grad_norm": 0.2094171643257141,
+ "learning_rate": 7.361260053619303e-06,
+ "loss": 0.4919305324554443,
+ "step": 3800
+ },
+ {
+ "epoch": 1.0241311123772496,
+ "grad_norm": 0.16315558552742004,
+ "learning_rate": 7.321045576407507e-06,
+ "loss": 0.5338080406188965,
+ "step": 3820
+ },
+ {
+ "epoch": 1.0294935817944164,
+ "grad_norm": 0.20310278236865997,
+ "learning_rate": 7.2808310991957104e-06,
+ "loss": 0.4789735794067383,
+ "step": 3840
+ },
+ {
+ "epoch": 1.034856051211583,
+ "grad_norm": 0.13879640400409698,
+ "learning_rate": 7.240616621983915e-06,
+ "loss": 0.49851651191711427,
+ "step": 3860
+ },
+ {
+ "epoch": 1.0402185206287495,
+ "grad_norm": 0.1722245216369629,
+ "learning_rate": 7.200402144772119e-06,
+ "loss": 0.5306562900543212,
+ "step": 3880
+ },
+ {
+ "epoch": 1.045580990045916,
+ "grad_norm": 0.1506664901971817,
+ "learning_rate": 7.160187667560322e-06,
+ "loss": 0.45285625457763673,
+ "step": 3900
+ },
+ {
+ "epoch": 1.0509434594630827,
+ "grad_norm": 0.204021617770195,
+ "learning_rate": 7.119973190348526e-06,
+ "loss": 0.5161935329437256,
+ "step": 3920
+ },
+ {
+ "epoch": 1.0563059288802494,
+ "grad_norm": 0.20319899916648865,
+ "learning_rate": 7.07975871313673e-06,
+ "loss": 0.4824995040893555,
+ "step": 3940
+ },
+ {
+ "epoch": 1.061668398297416,
+ "grad_norm": 0.19432441890239716,
+ "learning_rate": 7.0395442359249335e-06,
+ "loss": 0.5660453796386719,
+ "step": 3960
+ },
+ {
+ "epoch": 1.0670308677145826,
+ "grad_norm": 0.2576168477535248,
+ "learning_rate": 6.999329758713137e-06,
+ "loss": 0.4815997123718262,
+ "step": 3980
+ },
+ {
+ "epoch": 1.0723933371317491,
+ "grad_norm": 0.27557438611984253,
+ "learning_rate": 6.959115281501341e-06,
+ "loss": 0.43416056632995603,
+ "step": 4000
+ },
+ {
+ "epoch": 1.0777558065489157,
+ "grad_norm": 0.17039135098457336,
+ "learning_rate": 6.9189008042895446e-06,
+ "loss": 0.4980440139770508,
+ "step": 4020
+ },
+ {
+ "epoch": 1.0831182759660825,
+ "grad_norm": 0.2580510675907135,
+ "learning_rate": 6.878686327077748e-06,
+ "loss": 0.5068618774414062,
+ "step": 4040
+ },
+ {
+ "epoch": 1.088480745383249,
+ "grad_norm": 0.14738141000270844,
+ "learning_rate": 6.838471849865952e-06,
+ "loss": 0.4890751361846924,
+ "step": 4060
+ },
+ {
+ "epoch": 1.0938432148004156,
+ "grad_norm": 0.2081380933523178,
+ "learning_rate": 6.798257372654156e-06,
+ "loss": 0.5679311275482177,
+ "step": 4080
+ },
+ {
+ "epoch": 1.0992056842175821,
+ "grad_norm": 0.17693300545215607,
+ "learning_rate": 6.758042895442359e-06,
+ "loss": 0.5189684391021728,
+ "step": 4100
+ },
+ {
+ "epoch": 1.104568153634749,
+ "grad_norm": 0.23674148321151733,
+ "learning_rate": 6.717828418230563e-06,
+ "loss": 0.48049330711364746,
+ "step": 4120
+ },
+ {
+ "epoch": 1.1099306230519155,
+ "grad_norm": 0.21366719901561737,
+ "learning_rate": 6.677613941018767e-06,
+ "loss": 0.4967336654663086,
+ "step": 4140
+ },
+ {
+ "epoch": 1.115293092469082,
+ "grad_norm": 0.19616496562957764,
+ "learning_rate": 6.6373994638069704e-06,
+ "loss": 0.46569108963012695,
+ "step": 4160
+ },
+ {
+ "epoch": 1.1206555618862486,
+ "grad_norm": 0.17559197545051575,
+ "learning_rate": 6.597184986595174e-06,
+ "loss": 0.49478998184204104,
+ "step": 4180
+ },
+ {
+ "epoch": 1.1260180313034152,
+ "grad_norm": 0.184451162815094,
+ "learning_rate": 6.556970509383378e-06,
+ "loss": 0.5000570774078369,
+ "step": 4200
+ },
+ {
+ "epoch": 1.131380500720582,
+ "grad_norm": 0.18627093732357025,
+ "learning_rate": 6.5167560321715815e-06,
+ "loss": 0.5214301586151123,
+ "step": 4220
+ },
+ {
+ "epoch": 1.1367429701377485,
+ "grad_norm": 0.2080899477005005,
+ "learning_rate": 6.476541554959785e-06,
+ "loss": 0.47851176261901857,
+ "step": 4240
+ },
+ {
+ "epoch": 1.142105439554915,
+ "grad_norm": 0.18619345128536224,
+ "learning_rate": 6.436327077747989e-06,
+ "loss": 0.5022239685058594,
+ "step": 4260
+ },
+ {
+ "epoch": 1.1474679089720816,
+ "grad_norm": 0.23693107068538666,
+ "learning_rate": 6.396112600536193e-06,
+ "loss": 0.5198223114013671,
+ "step": 4280
+ },
+ {
+ "epoch": 1.1528303783892482,
+ "grad_norm": 0.17998561263084412,
+ "learning_rate": 6.355898123324397e-06,
+ "loss": 0.5228567123413086,
+ "step": 4300
+ },
+ {
+ "epoch": 1.158192847806415,
+ "grad_norm": 0.2783758342266083,
+ "learning_rate": 6.315683646112601e-06,
+ "loss": 0.5318965435028076,
+ "step": 4320
+ },
+ {
+ "epoch": 1.1635553172235815,
+ "grad_norm": 0.19693782925605774,
+ "learning_rate": 6.2754691689008046e-06,
+ "loss": 0.48392295837402344,
+ "step": 4340
+ },
+ {
+ "epoch": 1.168917786640748,
+ "grad_norm": 0.15940269827842712,
+ "learning_rate": 6.235254691689008e-06,
+ "loss": 0.4617619514465332,
+ "step": 4360
+ },
+ {
+ "epoch": 1.1742802560579146,
+ "grad_norm": 0.24782665073871613,
+ "learning_rate": 6.195040214477212e-06,
+ "loss": 0.49810285568237306,
+ "step": 4380
+ },
+ {
+ "epoch": 1.1796427254750812,
+ "grad_norm": 0.1946037858724594,
+ "learning_rate": 6.154825737265416e-06,
+ "loss": 0.4826976776123047,
+ "step": 4400
+ },
+ {
+ "epoch": 1.185005194892248,
+ "grad_norm": 0.16667844355106354,
+ "learning_rate": 6.114611260053619e-06,
+ "loss": 0.5159809589385986,
+ "step": 4420
+ },
+ {
+ "epoch": 1.1903676643094145,
+ "grad_norm": 0.19206570088863373,
+ "learning_rate": 6.074396782841823e-06,
+ "loss": 0.47541089057922364,
+ "step": 4440
+ },
+ {
+ "epoch": 1.195730133726581,
+ "grad_norm": 0.17394617199897766,
+ "learning_rate": 6.034182305630027e-06,
+ "loss": 0.5470661640167236,
+ "step": 4460
+ },
+ {
+ "epoch": 1.2010926031437477,
+ "grad_norm": 0.210404634475708,
+ "learning_rate": 5.993967828418231e-06,
+ "loss": 0.5377882957458496,
+ "step": 4480
+ },
+ {
+ "epoch": 1.2064550725609142,
+ "grad_norm": 0.18084648251533508,
+ "learning_rate": 5.953753351206435e-06,
+ "loss": 0.5037185192108155,
+ "step": 4500
+ },
+ {
+ "epoch": 1.211817541978081,
+ "grad_norm": 0.23707027733325958,
+ "learning_rate": 5.913538873994639e-06,
+ "loss": 0.4822190284729004,
+ "step": 4520
+ },
+ {
+ "epoch": 1.2171800113952476,
+ "grad_norm": 0.16474473476409912,
+ "learning_rate": 5.873324396782842e-06,
+ "loss": 0.46645288467407225,
+ "step": 4540
+ },
+ {
+ "epoch": 1.2225424808124141,
+ "grad_norm": 0.2142348438501358,
+ "learning_rate": 5.833109919571046e-06,
+ "loss": 0.5255855560302735,
+ "step": 4560
+ },
+ {
+ "epoch": 1.2279049502295807,
+ "grad_norm": 0.2531765103340149,
+ "learning_rate": 5.79289544235925e-06,
+ "loss": 0.507044792175293,
+ "step": 4580
+ },
+ {
+ "epoch": 1.2332674196467472,
+ "grad_norm": 0.2553550899028778,
+ "learning_rate": 5.7526809651474535e-06,
+ "loss": 0.4767824649810791,
+ "step": 4600
+ },
+ {
+ "epoch": 1.238629889063914,
+ "grad_norm": 0.14484412968158722,
+ "learning_rate": 5.712466487935657e-06,
+ "loss": 0.4675601005554199,
+ "step": 4620
+ },
+ {
+ "epoch": 1.2439923584810806,
+ "grad_norm": 0.14328251779079437,
+ "learning_rate": 5.672252010723861e-06,
+ "loss": 0.4956005573272705,
+ "step": 4640
+ },
+ {
+ "epoch": 1.2493548278982471,
+ "grad_norm": 0.1739245355129242,
+ "learning_rate": 5.632037533512065e-06,
+ "loss": 0.48583345413208007,
+ "step": 4660
+ },
+ {
+ "epoch": 1.2547172973154137,
+ "grad_norm": 0.21294184029102325,
+ "learning_rate": 5.591823056300268e-06,
+ "loss": 0.520921277999878,
+ "step": 4680
+ },
+ {
+ "epoch": 1.2600797667325803,
+ "grad_norm": 0.25132355093955994,
+ "learning_rate": 5.551608579088472e-06,
+ "loss": 0.5295385837554931,
+ "step": 4700
+ },
+ {
+ "epoch": 1.265442236149747,
+ "grad_norm": 0.18603841960430145,
+ "learning_rate": 5.511394101876676e-06,
+ "loss": 0.47570199966430665,
+ "step": 4720
+ },
+ {
+ "epoch": 1.2708047055669136,
+ "grad_norm": 0.19883134961128235,
+ "learning_rate": 5.471179624664879e-06,
+ "loss": 0.5016080379486084,
+ "step": 4740
+ },
+ {
+ "epoch": 1.2761671749840802,
+ "grad_norm": 0.19640181958675385,
+ "learning_rate": 5.430965147453083e-06,
+ "loss": 0.4999081134796143,
+ "step": 4760
+ },
+ {
+ "epoch": 1.2815296444012467,
+ "grad_norm": 0.2584764361381531,
+ "learning_rate": 5.390750670241287e-06,
+ "loss": 0.4780082702636719,
+ "step": 4780
+ },
+ {
+ "epoch": 1.2868921138184133,
+ "grad_norm": 0.2925741374492645,
+ "learning_rate": 5.3505361930294905e-06,
+ "loss": 0.5131395816802978,
+ "step": 4800
+ },
+ {
+ "epoch": 1.29225458323558,
+ "grad_norm": 0.18971531093120575,
+ "learning_rate": 5.310321715817694e-06,
+ "loss": 0.455674409866333,
+ "step": 4820
+ },
+ {
+ "epoch": 1.2976170526527466,
+ "grad_norm": 0.16778405010700226,
+ "learning_rate": 5.270107238605898e-06,
+ "loss": 0.5070962905883789,
+ "step": 4840
+ },
+ {
+ "epoch": 1.3029795220699132,
+ "grad_norm": 0.30026957392692566,
+ "learning_rate": 5.2298927613941016e-06,
+ "loss": 0.5120027542114258,
+ "step": 4860
+ },
+ {
+ "epoch": 1.3083419914870797,
+ "grad_norm": 0.17846634984016418,
+ "learning_rate": 5.189678284182305e-06,
+ "loss": 0.5114477157592774,
+ "step": 4880
+ },
+ {
+ "epoch": 1.3137044609042463,
+ "grad_norm": 0.1962418258190155,
+ "learning_rate": 5.149463806970509e-06,
+ "loss": 0.5043613910675049,
+ "step": 4900
+ },
+ {
+ "epoch": 1.319066930321413,
+ "grad_norm": 0.18446756899356842,
+ "learning_rate": 5.1092493297587135e-06,
+ "loss": 0.5396455287933349,
+ "step": 4920
+ },
+ {
+ "epoch": 1.3244293997385796,
+ "grad_norm": 0.20886844396591187,
+ "learning_rate": 5.069034852546917e-06,
+ "loss": 0.4879767417907715,
+ "step": 4940
+ },
+ {
+ "epoch": 1.3297918691557462,
+ "grad_norm": 0.16687901318073273,
+ "learning_rate": 5.028820375335121e-06,
+ "loss": 0.5014327049255372,
+ "step": 4960
+ },
+ {
+ "epoch": 1.3351543385729128,
+ "grad_norm": 0.19595153629779816,
+ "learning_rate": 4.988605898123325e-06,
+ "loss": 0.5375277996063232,
+ "step": 4980
+ },
+ {
+ "epoch": 1.3405168079900793,
+ "grad_norm": 0.2372344732284546,
+ "learning_rate": 4.948391420911528e-06,
+ "loss": 0.5020076274871826,
+ "step": 5000
+ },
+ {
+ "epoch": 1.345879277407246,
+ "grad_norm": 0.21030014753341675,
+ "learning_rate": 4.908176943699732e-06,
+ "loss": 0.5111066818237304,
+ "step": 5020
+ },
+ {
+ "epoch": 1.3512417468244127,
+ "grad_norm": 0.1866692751646042,
+ "learning_rate": 4.867962466487936e-06,
+ "loss": 0.4515383720397949,
+ "step": 5040
+ },
+ {
+ "epoch": 1.3566042162415792,
+ "grad_norm": 0.22531798481941223,
+ "learning_rate": 4.827747989276139e-06,
+ "loss": 0.4757690906524658,
+ "step": 5060
+ },
+ {
+ "epoch": 1.3619666856587458,
+ "grad_norm": 0.15868768095970154,
+ "learning_rate": 4.787533512064343e-06,
+ "loss": 0.45842318534851073,
+ "step": 5080
+ },
+ {
+ "epoch": 1.3673291550759124,
+ "grad_norm": 0.24528546631336212,
+ "learning_rate": 4.747319034852547e-06,
+ "loss": 0.47269258499145506,
+ "step": 5100
+ },
+ {
+ "epoch": 1.3726916244930791,
+ "grad_norm": 0.17387732863426208,
+ "learning_rate": 4.707104557640751e-06,
+ "loss": 0.5103805065155029,
+ "step": 5120
+ },
+ {
+ "epoch": 1.3780540939102457,
+ "grad_norm": 0.20686905086040497,
+ "learning_rate": 4.666890080428955e-06,
+ "loss": 0.5135180950164795,
+ "step": 5140
+ },
+ {
+ "epoch": 1.3834165633274123,
+ "grad_norm": 0.19599783420562744,
+ "learning_rate": 4.626675603217159e-06,
+ "loss": 0.5045839786529541,
+ "step": 5160
+ },
+ {
+ "epoch": 1.3887790327445788,
+ "grad_norm": 0.2585010528564453,
+ "learning_rate": 4.586461126005362e-06,
+ "loss": 0.45903496742248534,
+ "step": 5180
+ },
+ {
+ "epoch": 1.3941415021617454,
+ "grad_norm": 0.1688319593667984,
+ "learning_rate": 4.546246648793566e-06,
+ "loss": 0.5017509937286377,
+ "step": 5200
+ },
+ {
+ "epoch": 1.3995039715789122,
+ "grad_norm": 0.21520815789699554,
+ "learning_rate": 4.50603217158177e-06,
+ "loss": 0.48459539413452146,
+ "step": 5220
+ },
+ {
+ "epoch": 1.4048664409960787,
+ "grad_norm": 0.20514647662639618,
+ "learning_rate": 4.4658176943699735e-06,
+ "loss": 0.5073423862457276,
+ "step": 5240
+ },
+ {
+ "epoch": 1.4102289104132453,
+ "grad_norm": 0.21835413575172424,
+ "learning_rate": 4.425603217158177e-06,
+ "loss": 0.5290310382843018,
+ "step": 5260
+ },
+ {
+ "epoch": 1.4155913798304118,
+ "grad_norm": 0.28042587637901306,
+ "learning_rate": 4.385388739946381e-06,
+ "loss": 0.4823312759399414,
+ "step": 5280
+ },
+ {
+ "epoch": 1.4209538492475784,
+ "grad_norm": 0.18959026038646698,
+ "learning_rate": 4.345174262734585e-06,
+ "loss": 0.4921241760253906,
+ "step": 5300
+ },
+ {
+ "epoch": 1.4263163186647452,
+ "grad_norm": 0.18584316968917847,
+ "learning_rate": 4.304959785522788e-06,
+ "loss": 0.4892130374908447,
+ "step": 5320
+ },
+ {
+ "epoch": 1.4316787880819117,
+ "grad_norm": 0.17588038742542267,
+ "learning_rate": 4.264745308310992e-06,
+ "loss": 0.4822041988372803,
+ "step": 5340
+ },
+ {
+ "epoch": 1.4370412574990783,
+ "grad_norm": 0.18146033585071564,
+ "learning_rate": 4.224530831099196e-06,
+ "loss": 0.5084807395935058,
+ "step": 5360
+ },
+ {
+ "epoch": 1.4424037269162449,
+ "grad_norm": 0.2251797467470169,
+ "learning_rate": 4.184316353887399e-06,
+ "loss": 0.5146170139312745,
+ "step": 5380
+ },
+ {
+ "epoch": 1.4477661963334114,
+ "grad_norm": 0.18744796514511108,
+ "learning_rate": 4.144101876675603e-06,
+ "loss": 0.5189927577972412,
+ "step": 5400
+ },
+ {
+ "epoch": 1.4531286657505782,
+ "grad_norm": 0.25737133622169495,
+ "learning_rate": 4.103887399463807e-06,
+ "loss": 0.4891658782958984,
+ "step": 5420
+ },
+ {
+ "epoch": 1.4584911351677448,
+ "grad_norm": 0.20580479502677917,
+ "learning_rate": 4.0636729222520105e-06,
+ "loss": 0.4953591823577881,
+ "step": 5440
+ },
+ {
+ "epoch": 1.4638536045849113,
+ "grad_norm": 0.2351546287536621,
+ "learning_rate": 4.023458445040214e-06,
+ "loss": 0.5025320053100586,
+ "step": 5460
+ },
+ {
+ "epoch": 1.4692160740020779,
+ "grad_norm": 0.1819481998682022,
+ "learning_rate": 3.983243967828418e-06,
+ "loss": 0.47151756286621094,
+ "step": 5480
+ },
+ {
+ "epoch": 1.4745785434192444,
+ "grad_norm": 0.20772472023963928,
+ "learning_rate": 3.943029490616622e-06,
+ "loss": 0.4678915023803711,
+ "step": 5500
+ },
+ {
+ "epoch": 1.4799410128364112,
+ "grad_norm": 0.2203037440776825,
+ "learning_rate": 3.902815013404825e-06,
+ "loss": 0.46007452011108396,
+ "step": 5520
+ },
+ {
+ "epoch": 1.4853034822535778,
+ "grad_norm": 0.15371400117874146,
+ "learning_rate": 3.86260053619303e-06,
+ "loss": 0.44407024383544924,
+ "step": 5540
+ },
+ {
+ "epoch": 1.4906659516707443,
+ "grad_norm": 0.2276080846786499,
+ "learning_rate": 3.8223860589812335e-06,
+ "loss": 0.4730556488037109,
+ "step": 5560
+ },
+ {
+ "epoch": 1.4960284210879111,
+ "grad_norm": 0.24482466280460358,
+ "learning_rate": 3.7821715817694376e-06,
+ "loss": 0.5073911666870117,
+ "step": 5580
+ },
+ {
+ "epoch": 1.5013908905050775,
+ "grad_norm": 0.20438458025455475,
+ "learning_rate": 3.741957104557641e-06,
+ "loss": 0.46701641082763673,
+ "step": 5600
+ },
+ {
+ "epoch": 1.5067533599222442,
+ "grad_norm": 0.19854313135147095,
+ "learning_rate": 3.7017426273458446e-06,
+ "loss": 0.46309399604797363,
+ "step": 5620
+ },
+ {
+ "epoch": 1.5121158293394108,
+ "grad_norm": 0.18356069922447205,
+ "learning_rate": 3.6615281501340483e-06,
+ "loss": 0.503613805770874,
+ "step": 5640
+ },
+ {
+ "epoch": 1.5174782987565774,
+ "grad_norm": 0.2009744495153427,
+ "learning_rate": 3.621313672922252e-06,
+ "loss": 0.4765054225921631,
+ "step": 5660
+ },
+ {
+ "epoch": 1.5228407681737441,
+ "grad_norm": 0.3058745563030243,
+ "learning_rate": 3.5810991957104557e-06,
+ "loss": 0.5179148197174073,
+ "step": 5680
+ },
+ {
+ "epoch": 1.5282032375909105,
+ "grad_norm": 0.17671597003936768,
+ "learning_rate": 3.54088471849866e-06,
+ "loss": 0.45907344818115237,
+ "step": 5700
+ },
+ {
+ "epoch": 1.5335657070080773,
+ "grad_norm": 0.22209160029888153,
+ "learning_rate": 3.5006702412868635e-06,
+ "loss": 0.49304862022399903,
+ "step": 5720
+ },
+ {
+ "epoch": 1.5389281764252438,
+ "grad_norm": 0.21018914878368378,
+ "learning_rate": 3.4604557640750672e-06,
+ "loss": 0.5536758422851562,
+ "step": 5740
+ },
+ {
+ "epoch": 1.5442906458424104,
+ "grad_norm": 0.14339996874332428,
+ "learning_rate": 3.420241286863271e-06,
+ "loss": 0.48726091384887693,
+ "step": 5760
+ },
+ {
+ "epoch": 1.5496531152595772,
+ "grad_norm": 0.11419746279716492,
+ "learning_rate": 3.3800268096514746e-06,
+ "loss": 0.4514151573181152,
+ "step": 5780
+ },
+ {
+ "epoch": 1.5550155846767435,
+ "grad_norm": 0.18168962001800537,
+ "learning_rate": 3.3398123324396783e-06,
+ "loss": 0.5279990196228027,
+ "step": 5800
+ },
+ {
+ "epoch": 1.5603780540939103,
+ "grad_norm": 0.24244488775730133,
+ "learning_rate": 3.299597855227882e-06,
+ "loss": 0.49297361373901366,
+ "step": 5820
+ },
+ {
+ "epoch": 1.5657405235110768,
+ "grad_norm": 0.2017296999692917,
+ "learning_rate": 3.2593833780160857e-06,
+ "loss": 0.49305019378662107,
+ "step": 5840
+ },
+ {
+ "epoch": 1.5711029929282434,
+ "grad_norm": 0.22592377662658691,
+ "learning_rate": 3.2191689008042894e-06,
+ "loss": 0.4862989902496338,
+ "step": 5860
+ },
+ {
+ "epoch": 1.5764654623454102,
+ "grad_norm": 0.24772357940673828,
+ "learning_rate": 3.1789544235924935e-06,
+ "loss": 0.45182647705078127,
+ "step": 5880
+ },
+ {
+ "epoch": 1.5818279317625765,
+ "grad_norm": 0.20607218146324158,
+ "learning_rate": 3.1387399463806972e-06,
+ "loss": 0.48905248641967775,
+ "step": 5900
+ },
+ {
+ "epoch": 1.5871904011797433,
+ "grad_norm": 0.1931353509426117,
+ "learning_rate": 3.098525469168901e-06,
+ "loss": 0.5307461261749268,
+ "step": 5920
+ },
+ {
+ "epoch": 1.5925528705969099,
+ "grad_norm": 0.16020581126213074,
+ "learning_rate": 3.0583109919571046e-06,
+ "loss": 0.4672811985015869,
+ "step": 5940
+ },
+ {
+ "epoch": 1.5979153400140764,
+ "grad_norm": 0.23668015003204346,
+ "learning_rate": 3.0180965147453083e-06,
+ "loss": 0.5272688865661621,
+ "step": 5960
+ },
+ {
+ "epoch": 1.6032778094312432,
+ "grad_norm": 0.1916576772928238,
+ "learning_rate": 2.977882037533512e-06,
+ "loss": 0.4859332084655762,
+ "step": 5980
+ },
+ {
+ "epoch": 1.6086402788484095,
+ "grad_norm": 0.23635101318359375,
+ "learning_rate": 2.9376675603217157e-06,
+ "loss": 0.5418910980224609,
+ "step": 6000
+ },
+ {
+ "epoch": 1.6140027482655763,
+ "grad_norm": 0.2404562532901764,
+ "learning_rate": 2.89745308310992e-06,
+ "loss": 0.5449445247650146,
+ "step": 6020
+ },
+ {
+ "epoch": 1.6193652176827429,
+ "grad_norm": 0.20147347450256348,
+ "learning_rate": 2.8572386058981235e-06,
+ "loss": 0.4737790584564209,
+ "step": 6040
+ },
+ {
+ "epoch": 1.6247276870999094,
+ "grad_norm": 0.2455863654613495,
+ "learning_rate": 2.8170241286863272e-06,
+ "loss": 0.4722298145294189,
+ "step": 6060
+ },
+ {
+ "epoch": 1.6300901565170762,
+ "grad_norm": 0.22172148525714874,
+ "learning_rate": 2.776809651474531e-06,
+ "loss": 0.5120372295379638,
+ "step": 6080
+ },
+ {
+ "epoch": 1.6354526259342426,
+ "grad_norm": 0.3848462700843811,
+ "learning_rate": 2.7365951742627346e-06,
+ "loss": 0.5152206897735596,
+ "step": 6100
+ },
+ {
+ "epoch": 1.6408150953514093,
+ "grad_norm": 0.19071047008037567,
+ "learning_rate": 2.6963806970509383e-06,
+ "loss": 0.4757692813873291,
+ "step": 6120
+ },
+ {
+ "epoch": 1.646177564768576,
+ "grad_norm": 0.20568661391735077,
+ "learning_rate": 2.656166219839142e-06,
+ "loss": 0.475917387008667,
+ "step": 6140
+ },
+ {
+ "epoch": 1.6515400341857425,
+ "grad_norm": 0.11777322739362717,
+ "learning_rate": 2.6159517426273457e-06,
+ "loss": 0.5161296367645264,
+ "step": 6160
+ },
+ {
+ "epoch": 1.6569025036029092,
+ "grad_norm": 0.1700555831193924,
+ "learning_rate": 2.5757372654155494e-06,
+ "loss": 0.4715432167053223,
+ "step": 6180
+ },
+ {
+ "epoch": 1.6622649730200756,
+ "grad_norm": 0.18927083909511566,
+ "learning_rate": 2.5355227882037535e-06,
+ "loss": 0.49937710762023924,
+ "step": 6200
+ },
+ {
+ "epoch": 1.6676274424372424,
+ "grad_norm": 0.22097784280776978,
+ "learning_rate": 2.4953083109919572e-06,
+ "loss": 0.43366107940673826,
+ "step": 6220
+ },
+ {
+ "epoch": 1.672989911854409,
+ "grad_norm": 0.2299281805753708,
+ "learning_rate": 2.455093833780161e-06,
+ "loss": 0.5145821094512939,
+ "step": 6240
+ },
+ {
+ "epoch": 1.6783523812715755,
+ "grad_norm": 0.2384844720363617,
+ "learning_rate": 2.4148793565683646e-06,
+ "loss": 0.459308385848999,
+ "step": 6260
+ },
+ {
+ "epoch": 1.6837148506887423,
+ "grad_norm": 0.24471035599708557,
+ "learning_rate": 2.3746648793565683e-06,
+ "loss": 0.4676504611968994,
+ "step": 6280
+ },
+ {
+ "epoch": 1.6890773201059086,
+ "grad_norm": 0.24419866502285004,
+ "learning_rate": 2.334450402144772e-06,
+ "loss": 0.4745138168334961,
+ "step": 6300
+ },
+ {
+ "epoch": 1.6944397895230754,
+ "grad_norm": 0.15896575152873993,
+ "learning_rate": 2.294235924932976e-06,
+ "loss": 0.5073649883270264,
+ "step": 6320
+ },
+ {
+ "epoch": 1.699802258940242,
+ "grad_norm": 0.26504868268966675,
+ "learning_rate": 2.25402144772118e-06,
+ "loss": 0.4534353733062744,
+ "step": 6340
+ },
+ {
+ "epoch": 1.7051647283574085,
+ "grad_norm": 0.2461850792169571,
+ "learning_rate": 2.2138069705093836e-06,
+ "loss": 0.4862947940826416,
+ "step": 6360
+ },
+ {
+ "epoch": 1.7105271977745753,
+ "grad_norm": 0.17332817614078522,
+ "learning_rate": 2.1735924932975873e-06,
+ "loss": 0.5049370765686035,
+ "step": 6380
+ },
+ {
+ "epoch": 1.7158896671917419,
+ "grad_norm": 0.19762548804283142,
+ "learning_rate": 2.133378016085791e-06,
+ "loss": 0.5272616386413574,
+ "step": 6400
+ },
+ {
+ "epoch": 1.7212521366089084,
+ "grad_norm": 0.23265399038791656,
+ "learning_rate": 2.0931635388739946e-06,
+ "loss": 0.47600841522216797,
+ "step": 6420
+ },
+ {
+ "epoch": 1.726614606026075,
+ "grad_norm": 0.20868578553199768,
+ "learning_rate": 2.0529490616621983e-06,
+ "loss": 0.5027226448059082,
+ "step": 6440
+ },
+ {
+ "epoch": 1.7319770754432415,
+ "grad_norm": 0.2851981520652771,
+ "learning_rate": 2.012734584450402e-06,
+ "loss": 0.5288124561309815,
+ "step": 6460
+ },
+ {
+ "epoch": 1.7373395448604083,
+ "grad_norm": 0.20086587965488434,
+ "learning_rate": 1.9725201072386057e-06,
+ "loss": 0.4625516891479492,
+ "step": 6480
+ },
+ {
+ "epoch": 1.7427020142775749,
+ "grad_norm": 0.24060192704200745,
+ "learning_rate": 1.93230563002681e-06,
+ "loss": 0.4843903541564941,
+ "step": 6500
+ },
+ {
+ "epoch": 1.7480644836947414,
+ "grad_norm": 0.33561915159225464,
+ "learning_rate": 1.8920911528150133e-06,
+ "loss": 0.4823720932006836,
+ "step": 6520
+ },
+ {
+ "epoch": 1.753426953111908,
+ "grad_norm": 0.2510465383529663,
+ "learning_rate": 1.851876675603217e-06,
+ "loss": 0.46517143249511717,
+ "step": 6540
+ },
+ {
+ "epoch": 1.7587894225290746,
+ "grad_norm": 0.2631177604198456,
+ "learning_rate": 1.811662198391421e-06,
+ "loss": 0.5004732131958007,
+ "step": 6560
+ },
+ {
+ "epoch": 1.7641518919462413,
+ "grad_norm": 0.3493230640888214,
+ "learning_rate": 1.7714477211796249e-06,
+ "loss": 0.523811674118042,
+ "step": 6580
+ },
+ {
+ "epoch": 1.769514361363408,
+ "grad_norm": 0.1742691546678543,
+ "learning_rate": 1.7312332439678286e-06,
+ "loss": 0.5276295661926269,
+ "step": 6600
+ },
+ {
+ "epoch": 1.7748768307805745,
+ "grad_norm": 0.16134823858737946,
+ "learning_rate": 1.6910187667560323e-06,
+ "loss": 0.5352637290954589,
+ "step": 6620
+ },
+ {
+ "epoch": 1.780239300197741,
+ "grad_norm": 0.20977018773555756,
+ "learning_rate": 1.650804289544236e-06,
+ "loss": 0.4955774784088135,
+ "step": 6640
+ },
+ {
+ "epoch": 1.7856017696149076,
+ "grad_norm": 0.20511005818843842,
+ "learning_rate": 1.6105898123324397e-06,
+ "loss": 0.48643174171447756,
+ "step": 6660
+ },
+ {
+ "epoch": 1.7909642390320744,
+ "grad_norm": 0.23870044946670532,
+ "learning_rate": 1.5703753351206434e-06,
+ "loss": 0.4673162460327148,
+ "step": 6680
+ },
+ {
+ "epoch": 1.796326708449241,
+ "grad_norm": 0.21660065650939941,
+ "learning_rate": 1.5301608579088473e-06,
+ "loss": 0.5381903648376465,
+ "step": 6700
+ },
+ {
+ "epoch": 1.8016891778664075,
+ "grad_norm": 0.26977139711380005,
+ "learning_rate": 1.489946380697051e-06,
+ "loss": 0.42094998359680175,
+ "step": 6720
+ },
+ {
+ "epoch": 1.807051647283574,
+ "grad_norm": 0.2088550478219986,
+ "learning_rate": 1.4497319034852549e-06,
+ "loss": 0.49211792945861815,
+ "step": 6740
+ },
+ {
+ "epoch": 1.8124141167007406,
+ "grad_norm": 0.18141885101795197,
+ "learning_rate": 1.4095174262734586e-06,
+ "loss": 0.46572179794311525,
+ "step": 6760
+ },
+ {
+ "epoch": 1.8177765861179074,
+ "grad_norm": 0.2200685739517212,
+ "learning_rate": 1.3693029490616623e-06,
+ "loss": 0.4996177196502686,
+ "step": 6780
+ },
+ {
+ "epoch": 1.823139055535074,
+ "grad_norm": 0.19545452296733856,
+ "learning_rate": 1.329088471849866e-06,
+ "loss": 0.4731945514678955,
+ "step": 6800
+ },
+ {
+ "epoch": 1.8285015249522405,
+ "grad_norm": 0.2239731252193451,
+ "learning_rate": 1.2888739946380697e-06,
+ "loss": 0.47544050216674805,
+ "step": 6820
+ },
+ {
+ "epoch": 1.833863994369407,
+ "grad_norm": 0.22336581349372864,
+ "learning_rate": 1.2486595174262734e-06,
+ "loss": 0.47878737449645997,
+ "step": 6840
+ },
+ {
+ "epoch": 1.8392264637865736,
+ "grad_norm": 0.20921571552753448,
+ "learning_rate": 1.2084450402144773e-06,
+ "loss": 0.41347403526306153,
+ "step": 6860
+ },
+ {
+ "epoch": 1.8445889332037404,
+ "grad_norm": 0.1577194333076477,
+ "learning_rate": 1.168230563002681e-06,
+ "loss": 0.5464958667755127,
+ "step": 6880
+ },
+ {
+ "epoch": 1.849951402620907,
+ "grad_norm": 0.1477355808019638,
+ "learning_rate": 1.1280160857908849e-06,
+ "loss": 0.48548617362976076,
+ "step": 6900
+ },
+ {
+ "epoch": 1.8553138720380735,
+ "grad_norm": 0.22352682054042816,
+ "learning_rate": 1.0878016085790886e-06,
+ "loss": 0.4518588542938232,
+ "step": 6920
+ },
+ {
+ "epoch": 1.8606763414552403,
+ "grad_norm": 0.19822706282138824,
+ "learning_rate": 1.0475871313672923e-06,
+ "loss": 0.4190972805023193,
+ "step": 6940
+ },
+ {
+ "epoch": 1.8660388108724066,
+ "grad_norm": 0.20670010149478912,
+ "learning_rate": 1.007372654155496e-06,
+ "loss": 0.5045090675354004,
+ "step": 6960
+ },
+ {
+ "epoch": 1.8714012802895734,
+ "grad_norm": 0.2154514342546463,
+ "learning_rate": 9.671581769436997e-07,
+ "loss": 0.4472982883453369,
+ "step": 6980
+ },
+ {
+ "epoch": 1.87676374970674,
+ "grad_norm": 0.19451302289962769,
+ "learning_rate": 9.269436997319035e-07,
+ "loss": 0.4328409194946289,
+ "step": 7000
+ },
+ {
+ "epoch": 1.8821262191239065,
+ "grad_norm": 0.20980985462665558,
+ "learning_rate": 8.867292225201073e-07,
+ "loss": 0.43312845230102537,
+ "step": 7020
+ },
+ {
+ "epoch": 1.8874886885410733,
+ "grad_norm": 0.19927652180194855,
+ "learning_rate": 8.46514745308311e-07,
+ "loss": 0.5255829811096191,
+ "step": 7040
+ },
+ {
+ "epoch": 1.8928511579582397,
+ "grad_norm": 0.2869090437889099,
+ "learning_rate": 8.063002680965148e-07,
+ "loss": 0.5301108360290527,
+ "step": 7060
+ },
+ {
+ "epoch": 1.8982136273754064,
+ "grad_norm": 0.16662724316120148,
+ "learning_rate": 7.660857908847185e-07,
+ "loss": 0.48042588233947753,
+ "step": 7080
+ },
+ {
+ "epoch": 1.903576096792573,
+ "grad_norm": 0.21045279502868652,
+ "learning_rate": 7.258713136729223e-07,
+ "loss": 0.4943391799926758,
+ "step": 7100
+ },
+ {
+ "epoch": 1.9089385662097396,
+ "grad_norm": 0.18638195097446442,
+ "learning_rate": 6.85656836461126e-07,
+ "loss": 0.48406500816345216,
+ "step": 7120
+ },
+ {
+ "epoch": 1.9143010356269063,
+ "grad_norm": 0.15428832173347473,
+ "learning_rate": 6.454423592493298e-07,
+ "loss": 0.5084923267364502,
+ "step": 7140
+ },
+ {
+ "epoch": 1.9196635050440727,
+ "grad_norm": 0.2415294051170349,
+ "learning_rate": 6.052278820375336e-07,
+ "loss": 0.5026498317718506,
+ "step": 7160
+ },
+ {
+ "epoch": 1.9250259744612395,
+ "grad_norm": 0.23021087050437927,
+ "learning_rate": 5.650134048257373e-07,
+ "loss": 0.5294596195220947,
+ "step": 7180
+ },
+ {
+ "epoch": 1.930388443878406,
+ "grad_norm": 0.21689893305301666,
+ "learning_rate": 5.24798927613941e-07,
+ "loss": 0.4569683074951172,
+ "step": 7200
+ },
+ {
+ "epoch": 1.9357509132955726,
+ "grad_norm": 0.25458022952079773,
+ "learning_rate": 4.845844504021448e-07,
+ "loss": 0.4905412197113037,
+ "step": 7220
+ },
+ {
+ "epoch": 1.9411133827127394,
+ "grad_norm": 0.20990079641342163,
+ "learning_rate": 4.4436997319034854e-07,
+ "loss": 0.49599390029907225,
+ "step": 7240
+ },
+ {
+ "epoch": 1.9464758521299057,
+ "grad_norm": 0.2381497025489807,
+ "learning_rate": 4.041554959785523e-07,
+ "loss": 0.5149903774261475,
+ "step": 7260
+ },
+ {
+ "epoch": 1.9518383215470725,
+ "grad_norm": 0.29006749391555786,
+ "learning_rate": 3.6394101876675604e-07,
+ "loss": 0.5063531875610352,
+ "step": 7280
+ },
+ {
+ "epoch": 1.957200790964239,
+ "grad_norm": 0.20159471035003662,
+ "learning_rate": 3.237265415549598e-07,
+ "loss": 0.5062472343444824,
+ "step": 7300
+ },
+ {
+ "epoch": 1.9625632603814056,
+ "grad_norm": 0.21182367205619812,
+ "learning_rate": 2.8351206434316354e-07,
+ "loss": 0.49964118003845215,
+ "step": 7320
+ },
+ {
+ "epoch": 1.9679257297985724,
+ "grad_norm": 0.28606927394866943,
+ "learning_rate": 2.432975871313673e-07,
+ "loss": 0.48647170066833495,
+ "step": 7340
+ },
+ {
+ "epoch": 1.9732881992157387,
+ "grad_norm": 0.2779375910758972,
+ "learning_rate": 2.0308310991957104e-07,
+ "loss": 0.4816920280456543,
+ "step": 7360
+ },
+ {
+ "epoch": 1.9786506686329055,
+ "grad_norm": 0.20412451028823853,
+ "learning_rate": 1.628686327077748e-07,
+ "loss": 0.5177248477935791,
+ "step": 7380
+ },
+ {
+ "epoch": 1.984013138050072,
+ "grad_norm": 0.19861914217472076,
+ "learning_rate": 1.2265415549597854e-07,
+ "loss": 0.5265318870544433,
+ "step": 7400
+ },
+ {
+ "epoch": 1.9893756074672386,
+ "grad_norm": 0.1868918091058731,
+ "learning_rate": 8.24396782841823e-08,
+ "loss": 0.45395827293395996,
+ "step": 7420
+ },
+ {
+ "epoch": 1.9947380768844054,
+ "grad_norm": 0.17822134494781494,
+ "learning_rate": 4.2225201072386056e-08,
+ "loss": 0.4699725151062012,
+ "step": 7440
+ },
+ {
+ "epoch": 2.0,
+ "grad_norm": 0.2196798026561737,
+ "learning_rate": 2.0107238605898125e-09,
+ "loss": 0.46166625022888186,
+ "step": 7460
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": true
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 9.191746827630674e+17,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-7460/training_args.bin b/checkpoint-7460/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-7460/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/checkpoint-800/README.md b/checkpoint-800/README.md
new file mode 100644
index 0000000000000000000000000000000000000000..784b7ac4c5a67a69c6bacecded0e80dafb756fa6
--- /dev/null
+++ b/checkpoint-800/README.md
@@ -0,0 +1,206 @@
+---
+base_model: Qwen/Qwen2.5-14B
+library_name: peft
+pipeline_tag: text-generation
+tags:
+- base_model:adapter:Qwen/Qwen2.5-14B
+- lora
+---
+
+# Model Card for Model ID
+
+
+
+
+
+## Model Details
+
+### Model Description
+
+
+
+
+
+- **Developed by:** [More Information Needed]
+- **Funded by [optional]:** [More Information Needed]
+- **Shared by [optional]:** [More Information Needed]
+- **Model type:** [More Information Needed]
+- **Language(s) (NLP):** [More Information Needed]
+- **License:** [More Information Needed]
+- **Finetuned from model [optional]:** [More Information Needed]
+
+### Model Sources [optional]
+
+
+
+- **Repository:** [More Information Needed]
+- **Paper [optional]:** [More Information Needed]
+- **Demo [optional]:** [More Information Needed]
+
+## Uses
+
+
+
+### Direct Use
+
+
+
+[More Information Needed]
+
+### Downstream Use [optional]
+
+
+
+[More Information Needed]
+
+### Out-of-Scope Use
+
+
+
+[More Information Needed]
+
+## Bias, Risks, and Limitations
+
+
+
+[More Information Needed]
+
+### Recommendations
+
+
+
+Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations.
+
+## How to Get Started with the Model
+
+Use the code below to get started with the model.
+
+[More Information Needed]
+
+## Training Details
+
+### Training Data
+
+
+
+[More Information Needed]
+
+### Training Procedure
+
+
+
+#### Preprocessing [optional]
+
+[More Information Needed]
+
+
+#### Training Hyperparameters
+
+- **Training regime:** [More Information Needed]
+
+#### Speeds, Sizes, Times [optional]
+
+
+
+[More Information Needed]
+
+## Evaluation
+
+
+
+### Testing Data, Factors & Metrics
+
+#### Testing Data
+
+
+
+[More Information Needed]
+
+#### Factors
+
+
+
+[More Information Needed]
+
+#### Metrics
+
+
+
+[More Information Needed]
+
+### Results
+
+[More Information Needed]
+
+#### Summary
+
+
+
+## Model Examination [optional]
+
+
+
+[More Information Needed]
+
+## Environmental Impact
+
+
+
+Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700).
+
+- **Hardware Type:** [More Information Needed]
+- **Hours used:** [More Information Needed]
+- **Cloud Provider:** [More Information Needed]
+- **Compute Region:** [More Information Needed]
+- **Carbon Emitted:** [More Information Needed]
+
+## Technical Specifications [optional]
+
+### Model Architecture and Objective
+
+[More Information Needed]
+
+### Compute Infrastructure
+
+[More Information Needed]
+
+#### Hardware
+
+[More Information Needed]
+
+#### Software
+
+[More Information Needed]
+
+## Citation [optional]
+
+
+
+**BibTeX:**
+
+[More Information Needed]
+
+**APA:**
+
+[More Information Needed]
+
+## Glossary [optional]
+
+
+
+[More Information Needed]
+
+## More Information [optional]
+
+[More Information Needed]
+
+## Model Card Authors [optional]
+
+[More Information Needed]
+
+## Model Card Contact
+
+[More Information Needed]
+### Framework versions
+
+- PEFT 0.18.1
\ No newline at end of file
diff --git a/checkpoint-800/adapter_config.json b/checkpoint-800/adapter_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..fe26f7836e6cd73c1082af34b4d5921d1efb3d48
--- /dev/null
+++ b/checkpoint-800/adapter_config.json
@@ -0,0 +1,41 @@
+{
+ "alora_invocation_tokens": null,
+ "alpha_pattern": {},
+ "arrow_config": null,
+ "auto_mapping": null,
+ "base_model_name_or_path": null,
+ "bias": "none",
+ "corda_config": null,
+ "ensure_weight_tying": false,
+ "eva_config": null,
+ "exclude_modules": null,
+ "fan_in_fan_out": false,
+ "inference_mode": true,
+ "init_lora_weights": true,
+ "layer_replication": null,
+ "layers_pattern": null,
+ "layers_to_transform": null,
+ "loftq_config": {},
+ "lora_alpha": 32,
+ "lora_bias": false,
+ "lora_dropout": 0.05,
+ "megatron_config": null,
+ "megatron_core": "megatron.core",
+ "modules_to_save": null,
+ "peft_type": "LORA",
+ "peft_version": "0.18.1",
+ "qalora_group_size": 16,
+ "r": 16,
+ "rank_pattern": {},
+ "revision": null,
+ "target_modules": [
+ "v_proj",
+ "q_proj"
+ ],
+ "target_parameters": null,
+ "task_type": "CAUSAL_LM",
+ "trainable_token_indices": null,
+ "use_dora": false,
+ "use_qalora": false,
+ "use_rslora": false
+}
\ No newline at end of file
diff --git a/checkpoint-800/adapter_model.safetensors b/checkpoint-800/adapter_model.safetensors
new file mode 100644
index 0000000000000000000000000000000000000000..428fbbaae23118003b5f4ebfa7a79f2ab552e0c8
--- /dev/null
+++ b/checkpoint-800/adapter_model.safetensors
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b62aa4030982e8e3169a6adbec2f7a45ebe68eea7f45e8b730b8656e9414ac52
+size 50360752
diff --git a/checkpoint-800/chat_template.jinja b/checkpoint-800/chat_template.jinja
new file mode 100644
index 0000000000000000000000000000000000000000..28028c056af412405debd878cdda0171e35fa5d1
--- /dev/null
+++ b/checkpoint-800/chat_template.jinja
@@ -0,0 +1,54 @@
+{%- if tools %}
+ {{- '<|im_start|>system\n' }}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- messages[0]['content'] }}
+ {%- else %}
+ {{- 'You are a helpful assistant.' }}
+ {%- endif %}
+ {{- "\n\n# Tools\n\nYou may call one or more functions to assist with the user query.\n\nYou are provided with function signatures within XML tags:\n" }}
+ {%- for tool in tools %}
+ {{- "\n" }}
+ {{- tool | tojson }}
+ {%- endfor %}
+ {{- "\n\n\nFor each function call, return a json object with function name and arguments within XML tags:\n\n{\"name\": , \"arguments\": }\n<|im_end|>\n" }}
+{%- else %}
+ {%- if messages[0]['role'] == 'system' %}
+ {{- '<|im_start|>system\n' + messages[0]['content'] + '<|im_end|>\n' }}
+ {%- else %}
+ {{- '<|im_start|>system\nYou are a helpful assistant.<|im_end|>\n' }}
+ {%- endif %}
+{%- endif %}
+{%- for message in messages %}
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) or (message.role == "assistant" and not message.tool_calls) %}
+ {{- '<|im_start|>' + message.role + '\n' + message.content + '<|im_end|>' + '\n' }}
+ {%- elif message.role == "assistant" %}
+ {{- '<|im_start|>' + message.role }}
+ {%- if message.content %}
+ {{- '\n' + message.content }}
+ {%- endif %}
+ {%- for tool_call in message.tool_calls %}
+ {%- if tool_call.function is defined %}
+ {%- set tool_call = tool_call.function %}
+ {%- endif %}
+ {{- '\n\n{"name": "' }}
+ {{- tool_call.name }}
+ {{- '", "arguments": ' }}
+ {{- tool_call.arguments | tojson }}
+ {{- '}\n' }}
+ {%- endfor %}
+ {{- '<|im_end|>\n' }}
+ {%- elif message.role == "tool" %}
+ {%- if (loop.index0 == 0) or (messages[loop.index0 - 1].role != "tool") %}
+ {{- '<|im_start|>user' }}
+ {%- endif %}
+ {{- '\n\n' }}
+ {{- message.content }}
+ {{- '\n' }}
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
+ {{- '<|im_end|>\n' }}
+ {%- endif %}
+ {%- endif %}
+{%- endfor %}
+{%- if add_generation_prompt %}
+ {{- '<|im_start|>assistant\n' }}
+{%- endif %}
diff --git a/checkpoint-800/optimizer.pt b/checkpoint-800/optimizer.pt
new file mode 100644
index 0000000000000000000000000000000000000000..47d0cc019ff07ea57013a8ce804948317ed6166a
--- /dev/null
+++ b/checkpoint-800/optimizer.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a60aceea63941b6f07d7d1b9d66cfdfe24c0bd021d2f9f656d097cc8b57e327b
+size 100828235
diff --git a/checkpoint-800/rng_state.pth b/checkpoint-800/rng_state.pth
new file mode 100644
index 0000000000000000000000000000000000000000..a76e5716169bf8900f8577fd27094187888a25e6
--- /dev/null
+++ b/checkpoint-800/rng_state.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:63dd7726a052e5bc7ad80bc87b567a2ece2f71b63cea73b5c0f39c143f56dfe5
+size 14645
diff --git a/checkpoint-800/scheduler.pt b/checkpoint-800/scheduler.pt
new file mode 100644
index 0000000000000000000000000000000000000000..bbdce98612b26bd9935d387661e0d77ab2b596bc
--- /dev/null
+++ b/checkpoint-800/scheduler.pt
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c770d5e2f41bbfe7a718e461fc6c35726d183615e87d7855afe86b2ceea37a85
+size 1465
diff --git a/checkpoint-800/tokenizer.json b/checkpoint-800/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/checkpoint-800/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/checkpoint-800/tokenizer_config.json b/checkpoint-800/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/checkpoint-800/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/checkpoint-800/trainer_state.json b/checkpoint-800/trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..111742b7c8a060594898a67337614563824e088e
--- /dev/null
+++ b/checkpoint-800/trainer_state.json
@@ -0,0 +1,314 @@
+{
+ "best_global_step": null,
+ "best_metric": null,
+ "best_model_checkpoint": null,
+ "epoch": 0.2144987766866642,
+ "eval_steps": 500,
+ "global_step": 800,
+ "is_hyper_param_search": false,
+ "is_local_process_zero": true,
+ "is_world_process_zero": true,
+ "log_history": [
+ {
+ "epoch": 0.005362469417166605,
+ "grad_norm": 0.050072263926267624,
+ "learning_rate": 1.4961796246648793e-05,
+ "loss": 1.0673207283020019,
+ "step": 20
+ },
+ {
+ "epoch": 0.01072493883433321,
+ "grad_norm": 0.06825340539216995,
+ "learning_rate": 1.4921581769436997e-05,
+ "loss": 0.9185627937316895,
+ "step": 40
+ },
+ {
+ "epoch": 0.016087408251499815,
+ "grad_norm": 0.06827432662248611,
+ "learning_rate": 1.48813672922252e-05,
+ "loss": 0.7999343872070312,
+ "step": 60
+ },
+ {
+ "epoch": 0.02144987766866642,
+ "grad_norm": 0.05807405710220337,
+ "learning_rate": 1.4841152815013404e-05,
+ "loss": 0.7322770595550537,
+ "step": 80
+ },
+ {
+ "epoch": 0.026812347085833025,
+ "grad_norm": 0.06654328852891922,
+ "learning_rate": 1.4800938337801608e-05,
+ "loss": 0.7097890377044678,
+ "step": 100
+ },
+ {
+ "epoch": 0.03217481650299963,
+ "grad_norm": 0.09104783087968826,
+ "learning_rate": 1.4760723860589812e-05,
+ "loss": 0.6513629913330078,
+ "step": 120
+ },
+ {
+ "epoch": 0.03753728592016624,
+ "grad_norm": 0.10718850791454315,
+ "learning_rate": 1.4720509383378015e-05,
+ "loss": 0.678717851638794,
+ "step": 140
+ },
+ {
+ "epoch": 0.04289975533733284,
+ "grad_norm": 0.09187154471874237,
+ "learning_rate": 1.4680294906166219e-05,
+ "loss": 0.647278118133545,
+ "step": 160
+ },
+ {
+ "epoch": 0.04826222475449945,
+ "grad_norm": 0.07148946076631546,
+ "learning_rate": 1.4640080428954423e-05,
+ "loss": 0.6737877368927002,
+ "step": 180
+ },
+ {
+ "epoch": 0.05362469417166605,
+ "grad_norm": 0.08909227699041367,
+ "learning_rate": 1.4599865951742626e-05,
+ "loss": 0.6373191356658936,
+ "step": 200
+ },
+ {
+ "epoch": 0.05898716358883266,
+ "grad_norm": 0.07850278168916702,
+ "learning_rate": 1.455965147453083e-05,
+ "loss": 0.6020126819610596,
+ "step": 220
+ },
+ {
+ "epoch": 0.06434963300599926,
+ "grad_norm": 0.09538089483976364,
+ "learning_rate": 1.4519436997319034e-05,
+ "loss": 0.6096773147583008,
+ "step": 240
+ },
+ {
+ "epoch": 0.06971210242316586,
+ "grad_norm": 0.07478228211402893,
+ "learning_rate": 1.447922252010724e-05,
+ "loss": 0.6299086093902588,
+ "step": 260
+ },
+ {
+ "epoch": 0.07507457184033248,
+ "grad_norm": 0.1514953374862671,
+ "learning_rate": 1.4439008042895443e-05,
+ "loss": 0.5591042518615723,
+ "step": 280
+ },
+ {
+ "epoch": 0.08043704125749908,
+ "grad_norm": 0.08260886371135712,
+ "learning_rate": 1.4398793565683647e-05,
+ "loss": 0.6200376987457276,
+ "step": 300
+ },
+ {
+ "epoch": 0.08579951067466568,
+ "grad_norm": 0.17698714137077332,
+ "learning_rate": 1.435857908847185e-05,
+ "loss": 0.6023219585418701,
+ "step": 320
+ },
+ {
+ "epoch": 0.0911619800918323,
+ "grad_norm": 0.06104859337210655,
+ "learning_rate": 1.4318364611260054e-05,
+ "loss": 0.6181454658508301,
+ "step": 340
+ },
+ {
+ "epoch": 0.0965244495089989,
+ "grad_norm": 0.04990549385547638,
+ "learning_rate": 1.4278150134048258e-05,
+ "loss": 0.5593632698059082,
+ "step": 360
+ },
+ {
+ "epoch": 0.1018869189261655,
+ "grad_norm": 0.09426380693912506,
+ "learning_rate": 1.4237935656836461e-05,
+ "loss": 0.5790591716766358,
+ "step": 380
+ },
+ {
+ "epoch": 0.1072493883433321,
+ "grad_norm": 0.08783263713121414,
+ "learning_rate": 1.4197721179624665e-05,
+ "loss": 0.585063886642456,
+ "step": 400
+ },
+ {
+ "epoch": 0.11261185776049872,
+ "grad_norm": 0.06869607418775558,
+ "learning_rate": 1.4157506702412869e-05,
+ "loss": 0.5638764381408692,
+ "step": 420
+ },
+ {
+ "epoch": 0.11797432717766532,
+ "grad_norm": 0.10537438839673996,
+ "learning_rate": 1.4117292225201072e-05,
+ "loss": 0.6060166835784913,
+ "step": 440
+ },
+ {
+ "epoch": 0.12333679659483192,
+ "grad_norm": 0.09851580113172531,
+ "learning_rate": 1.4077077747989278e-05,
+ "loss": 0.5605969905853272,
+ "step": 460
+ },
+ {
+ "epoch": 0.12869926601199852,
+ "grad_norm": 0.11954096704721451,
+ "learning_rate": 1.4036863270777482e-05,
+ "loss": 0.5549856662750244,
+ "step": 480
+ },
+ {
+ "epoch": 0.13406173542916514,
+ "grad_norm": 0.13259431719779968,
+ "learning_rate": 1.3996648793565685e-05,
+ "loss": 0.5893547534942627,
+ "step": 500
+ },
+ {
+ "epoch": 0.13942420484633172,
+ "grad_norm": 0.11842650175094604,
+ "learning_rate": 1.3956434316353889e-05,
+ "loss": 0.6237683773040772,
+ "step": 520
+ },
+ {
+ "epoch": 0.14478667426349834,
+ "grad_norm": 0.1204022690653801,
+ "learning_rate": 1.3916219839142093e-05,
+ "loss": 0.572803258895874,
+ "step": 540
+ },
+ {
+ "epoch": 0.15014914368066495,
+ "grad_norm": 0.1345946341753006,
+ "learning_rate": 1.3876005361930296e-05,
+ "loss": 0.5632933139801025,
+ "step": 560
+ },
+ {
+ "epoch": 0.15551161309783154,
+ "grad_norm": 0.11733393371105194,
+ "learning_rate": 1.38357908847185e-05,
+ "loss": 0.6197309494018555,
+ "step": 580
+ },
+ {
+ "epoch": 0.16087408251499816,
+ "grad_norm": 0.0731734186410904,
+ "learning_rate": 1.3795576407506704e-05,
+ "loss": 0.5823808670043945,
+ "step": 600
+ },
+ {
+ "epoch": 0.16623655193216477,
+ "grad_norm": 0.09452618658542633,
+ "learning_rate": 1.3755361930294907e-05,
+ "loss": 0.5599356651306152,
+ "step": 620
+ },
+ {
+ "epoch": 0.17159902134933136,
+ "grad_norm": 0.09183815121650696,
+ "learning_rate": 1.3715147453083111e-05,
+ "loss": 0.5465828895568847,
+ "step": 640
+ },
+ {
+ "epoch": 0.17696149076649798,
+ "grad_norm": 0.0953364372253418,
+ "learning_rate": 1.3674932975871315e-05,
+ "loss": 0.5516108989715576,
+ "step": 660
+ },
+ {
+ "epoch": 0.1823239601836646,
+ "grad_norm": 0.11190114170312881,
+ "learning_rate": 1.3634718498659519e-05,
+ "loss": 0.5717048645019531,
+ "step": 680
+ },
+ {
+ "epoch": 0.18768642960083118,
+ "grad_norm": 0.11502158641815186,
+ "learning_rate": 1.3594504021447722e-05,
+ "loss": 0.528355598449707,
+ "step": 700
+ },
+ {
+ "epoch": 0.1930488990179978,
+ "grad_norm": 0.12480133026838303,
+ "learning_rate": 1.3554289544235926e-05,
+ "loss": 0.5860391616821289,
+ "step": 720
+ },
+ {
+ "epoch": 0.19841136843516438,
+ "grad_norm": 0.14408785104751587,
+ "learning_rate": 1.351407506702413e-05,
+ "loss": 0.5422697544097901,
+ "step": 740
+ },
+ {
+ "epoch": 0.203773837852331,
+ "grad_norm": 0.12405668199062347,
+ "learning_rate": 1.3473860589812333e-05,
+ "loss": 0.5876667499542236,
+ "step": 760
+ },
+ {
+ "epoch": 0.2091363072694976,
+ "grad_norm": 0.12171291559934616,
+ "learning_rate": 1.3433646112600537e-05,
+ "loss": 0.563751220703125,
+ "step": 780
+ },
+ {
+ "epoch": 0.2144987766866642,
+ "grad_norm": 0.10827518254518509,
+ "learning_rate": 1.339343163538874e-05,
+ "loss": 0.5700247764587403,
+ "step": 800
+ }
+ ],
+ "logging_steps": 20,
+ "max_steps": 7460,
+ "num_input_tokens_seen": 0,
+ "num_train_epochs": 2,
+ "save_steps": 200,
+ "stateful_callbacks": {
+ "TrainerControl": {
+ "args": {
+ "should_epoch_stop": false,
+ "should_evaluate": false,
+ "should_log": false,
+ "should_save": true,
+ "should_training_stop": false
+ },
+ "attributes": {}
+ }
+ },
+ "total_flos": 9.803862124388966e+16,
+ "train_batch_size": 1,
+ "trial_name": null,
+ "trial_params": null
+}
diff --git a/checkpoint-800/training_args.bin b/checkpoint-800/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/checkpoint-800/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201
diff --git a/tokenizer.json b/tokenizer.json
new file mode 100644
index 0000000000000000000000000000000000000000..e741ca70ace7c8d66f6ae643c234b1dbec9a0bfe
--- /dev/null
+++ b/tokenizer.json
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:21e2b58ce119ac9c0d306b7a35d538fe02f55e7f2af95cb0a2d563e892790684
+size 11421991
diff --git a/tokenizer_config.json b/tokenizer_config.json
new file mode 100644
index 0000000000000000000000000000000000000000..c960ecf0d33fd7b8c99d12680c0e74a82b36d446
--- /dev/null
+++ b/tokenizer_config.json
@@ -0,0 +1,29 @@
+{
+ "add_prefix_space": false,
+ "backend": "tokenizers",
+ "bos_token": null,
+ "clean_up_tokenization_spaces": false,
+ "eos_token": "<|endoftext|>",
+ "errors": "replace",
+ "extra_special_tokens": [
+ "<|im_start|>",
+ "<|im_end|>",
+ "<|object_ref_start|>",
+ "<|object_ref_end|>",
+ "<|box_start|>",
+ "<|box_end|>",
+ "<|quad_start|>",
+ "<|quad_end|>",
+ "<|vision_start|>",
+ "<|vision_end|>",
+ "<|vision_pad|>",
+ "<|image_pad|>",
+ "<|video_pad|>"
+ ],
+ "is_local": false,
+ "model_max_length": 131072,
+ "pad_token": "<|endoftext|>",
+ "split_special_tokens": false,
+ "tokenizer_class": "Qwen2Tokenizer",
+ "unk_token": null
+}
diff --git a/training_args.bin b/training_args.bin
new file mode 100644
index 0000000000000000000000000000000000000000..7c9b16244c86dffd05083c502a805fd59a32054c
--- /dev/null
+++ b/training_args.bin
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a01066b2f53606b4b364ae06eb8d2749e4ba60cb0815f7958c3b0381dfb4b1f4
+size 5201