spitfire4794 commited on
Commit
e61835a
·
verified ·
1 Parent(s): e544d05

Training in progress, step 500

Browse files
README.md ADDED
@@ -0,0 +1,58 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ base_model: SurjoLabs/Blaze-SFT
3
+ library_name: transformers
4
+ model_name: Blaze-Title
5
+ tags:
6
+ - generated_from_trainer
7
+ - trl
8
+ - sft
9
+ licence: license
10
+ ---
11
+
12
+ # Model Card for Blaze-Title
13
+
14
+ This model is a fine-tuned version of [SurjoLabs/Blaze-SFT](https://huggingface.co/SurjoLabs/Blaze-SFT).
15
+ It has been trained using [TRL](https://github.com/huggingface/trl).
16
+
17
+ ## Quick start
18
+
19
+ ```python
20
+ from transformers import pipeline
21
+
22
+ question = "If you had a time machine, but could only go to the past or the future once and never return, which would you choose and why?"
23
+ generator = pipeline("text-generation", model="SurjoLabs/Blaze-Title", device="cuda")
24
+ output = generator([{"role": "user", "content": question}], max_new_tokens=128, return_full_text=False)[0]
25
+ print(output["generated_text"])
26
+ ```
27
+
28
+ ## Training procedure
29
+
30
+
31
+
32
+
33
+
34
+ This model was trained with SFT.
35
+
36
+ ### Framework versions
37
+
38
+ - TRL: 1.12.0
39
+ - Transformers: 5.16.1
40
+ - Pytorch: 2.14.0
41
+ - Datasets: 5.0.1
42
+ - Tokenizers: 0.23.2
43
+
44
+ ## Citations
45
+
46
+
47
+
48
+ Cite TRL as:
49
+
50
+ ```bibtex
51
+ @software{vonwerra2020trl,
52
+ title = {{TRL: Transformers Reinforcement Learning}},
53
+ author = {von Werra, Leandro and Belkada, Younes and Tunstall, Lewis and Beeching, Edward and Thrush, Tristan and Lambert, Nathan and Huang, Shengyi and Rasul, Kashif and Gallouédec, Quentin},
54
+ license = {Apache-2.0},
55
+ url = {https://github.com/huggingface/trl},
56
+ year = {2020}
57
+ }
58
+ ```
chat_template.jinja ADDED
@@ -0,0 +1 @@
 
 
1
+ {{ bos_token }}{% for message in messages %}{% if message['role'] == 'assistant' %}{{ '<|im_start|>assistant\n' }}{% generation %}{{ message['content'] + '<|im_end|>\n' }}{% endgeneration %}{% else %}{{ '<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>\n' }}{% endif %}{% endfor %}{% if add_generation_prompt %}{{ '<|im_start|>assistant\n' }}{% endif %}
config.json ADDED
@@ -0,0 +1,44 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "BlazeForCausalLM"
4
+ ],
5
+ "attention_bias": false,
6
+ "attention_dropout": 0.0,
7
+ "auto_map": {
8
+ "AutoConfig": "configuration_blaze.BlazeConfig",
9
+ "AutoModelForCausalLM": "modeling_blaze.BlazeForCausalLM"
10
+ },
11
+ "bos_token_id": 2,
12
+ "coda_layers": 1,
13
+ "dtype": "bfloat16",
14
+ "eos_token_id": 6,
15
+ "gradient_checkpointing": false,
16
+ "head_dim": 64,
17
+ "hidden_act": "silu",
18
+ "hidden_size": 512,
19
+ "initializer_range": 0.02,
20
+ "intermediate_size": 1536,
21
+ "max_position_embeddings": 2048,
22
+ "mlp_bias": false,
23
+ "model_type": "blaze",
24
+ "num_attention_heads": 8,
25
+ "num_hidden_layers": 14,
26
+ "num_key_value_heads": 4,
27
+ "pad_token_id": 1,
28
+ "prelude_layers": 1,
29
+ "pretraining_tp": 1,
30
+ "recurrent_layers": 12,
31
+ "recurrent_passes": 2,
32
+ "rms_norm_eps": 1e-05,
33
+ "rope_parameters": {
34
+ "rope_theta": 10000.0,
35
+ "rope_type": "default"
36
+ },
37
+ "rope_theta": 10000.0,
38
+ "tie_word_embeddings": true,
39
+ "transformers_version": "5.16.1",
40
+ "use_cache": false,
41
+ "use_flash_attn": false,
42
+ "vocab_size": 8192,
43
+ "xsa_projection": true
44
+ }
generation_config.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_from_model_config": true,
3
+ "bos_token_id": 2,
4
+ "eos_token_id": [
5
+ 3,
6
+ 2,
7
+ 6
8
+ ],
9
+ "output_attentions": false,
10
+ "output_hidden_states": false,
11
+ "pad_token_id": 1,
12
+ "transformers_version": "5.16.1",
13
+ "use_cache": true
14
+ }
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1b55a7a80924d89cb1a5fa681af2eec6f748207fed02202bce3fb0f9787aef94
3
+ size 96519360
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1,85 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "backend": "tokenizers",
3
+ "bos_token": "<|bos|>",
4
+ "clean_up_tokenization_spaces": false,
5
+ "eos_token": "<|im_end|>",
6
+ "extra_special_tokens": [
7
+ "<|unk|>",
8
+ "<|pad|>",
9
+ "<|bos|>",
10
+ "<|eos|>",
11
+ "<|mask|>",
12
+ "<|im_start|>",
13
+ "<|im_end|>",
14
+ "<|system|>",
15
+ "<|user|>",
16
+ "<|assistant|>",
17
+ "<think>",
18
+ "</think>",
19
+ "<|begin_of_thought|>",
20
+ "<|end_of_thought|>",
21
+ "<answer>",
22
+ "</answer>",
23
+ "<|step|>",
24
+ "<|/step|>",
25
+ "<context>",
26
+ "</context>",
27
+ "<|doc_start|>",
28
+ "<|doc_end|>",
29
+ "<|search|>",
30
+ "<|search_results|>",
31
+ "<|tool_list_start|>",
32
+ "<|tool_list_end|>",
33
+ "<tools>",
34
+ "</tools>",
35
+ "<|tool_call_start|>",
36
+ "<|tool_call_end|>",
37
+ "<|tool_call|>",
38
+ "<|/tool_call|>",
39
+ "<|tool_response_start|>",
40
+ "<|tool_response_end|>",
41
+ "<|tool_response|>",
42
+ "<|/tool_response|>",
43
+ "<|image|>",
44
+ "<|image_pad|>",
45
+ "<|image_placeholder|>",
46
+ "<|audio|>",
47
+ "<|audio_pad|>",
48
+ "<|audio_placeholder|>",
49
+ "<|video|>",
50
+ "<|video_pad|>",
51
+ "<|fim_prefix|>",
52
+ "<|fim_suffix|>",
53
+ "<|fim_middle|>",
54
+ "<|repo_name|>",
55
+ "<|file_separator|>",
56
+ "<|reward|>",
57
+ "<|reserved_0|>",
58
+ "<|reserved_1|>",
59
+ "<|reserved_2|>",
60
+ "<|reserved_3|>",
61
+ "<|reserved_4|>",
62
+ "<|reserved_5|>",
63
+ "<|reserved_6|>",
64
+ "<|reserved_7|>",
65
+ "<|reserved_8|>",
66
+ "<|reserved_9|>",
67
+ "<|reserved_10|>",
68
+ "<|reserved_11|>",
69
+ "<|reserved_12|>",
70
+ "<|reserved_13|>",
71
+ "<|reserved_14|>",
72
+ "<|reserved_15|>",
73
+ "<|reserved_16|>",
74
+ "<|reserved_17|>",
75
+ "<|reserved_18|>",
76
+ "<|reserved_19|>"
77
+ ],
78
+ "is_local": false,
79
+ "local_files_only": false,
80
+ "mask_token": "<|mask|>",
81
+ "model_max_length": 10000000,
82
+ "pad_token": "<|pad|>",
83
+ "tokenizer_class": "TokenizersBackend",
84
+ "unk_token": "<|unk|>"
85
+ }
training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e03aae144474a03f8ecb82b049a21b4dd10f88c0bdcf9a88d1d04379f9c18f32
3
+ size 5841