luisastellet commited on
Commit
f1c4bb9
·
verified ·
1 Parent(s): 5a34b3a

tokenizer modelo_melhor_hp

Browse files
Files changed (3) hide show
  1. chat_template.jinja +83 -0
  2. tokenizer.json +0 -0
  3. tokenizer_config.json +27 -0
chat_template.jinja ADDED
@@ -0,0 +1,83 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {#- Handle tool/function calling setup #}
2
+ {%- if tools %}
3
+ {{- '<|im_start|>system\n' }}
4
+ {#- Include system message if present #}
5
+ {%- if messages[0].role == 'system' %}
6
+ {{- messages[0].content + '\n\n' }}
7
+ {%- endif %}
8
+ {#- Add tool calling instructions in Portuguese #}
9
+ {{- "# Tools / Ferramentas\n\nVocê pode chamar uma ou mais funções para auxiliar na consulta do usuário.\n\nVocê recebe assinaturas de funções dentro de tags XML <tools></tools>:\n<tools>" }}
10
+ {%- for tool in tools %}
11
+ {{- "\n" }}
12
+ {{- tool | tojson }}
13
+ {%- endfor %}
14
+ {{- "\n</tools>\n\nPara cada chamada de função, retorne um objeto json com o nome da função e os argumentos dentro das tags XML <tool_call></tool_call>:\n<tool_call>\n{\"name\": <function-name>, \"arguments\": <args-json-object>}\n</tool_call><|im_end|>\n" }}
15
+ {%- else %}
16
+ {#- Standard system message without tools #}
17
+ {%- if messages[0].role == 'system' %}
18
+ {{- '<|im_start|>system\n' + messages[0].content + '<|im_end|>\n' }}
19
+ {%- endif %}
20
+ {%- endif %}
21
+
22
+ {#- Process each message in the conversation #}
23
+ {%- for message in messages %}
24
+ {#- Normalize content to string #}
25
+ {%- if message.content is string %}
26
+ {%- set content = message.content %}
27
+ {%- else %}
28
+ {%- set content = '' %}
29
+ {%- endif %}
30
+
31
+ {#- Handle user messages and non-first system messages #}
32
+ {%- if (message.role == "user") or (message.role == "system" and not loop.first) %}
33
+ {{- '<|im_start|>' + message.role + '\n' + content + '<|im_end|>' + '\n' }}
34
+
35
+ {#- Handle assistant messages without reasoning #}
36
+ {%- elif message.role == "assistant" %}
37
+ {{- '<|im_start|>' + message.role }}
38
+ {% generation %}
39
+ {{- content }}
40
+
41
+ {#- Add tool calls if present #}
42
+ {%- if message.tool_calls %}
43
+ {%- for tool_call in message.tool_calls %}
44
+ {%- if (loop.first and content) or (not loop.first) %}
45
+ {{- '\n' }}
46
+ {%- endif %}
47
+ {#- Normalize tool call format #}
48
+ {%- if tool_call.function %}
49
+ {%- set tool_call = tool_call.function %}
50
+ {%- endif %}
51
+ {{- '<tool_call>\n{"name": "' }}
52
+ {{- tool_call.name }}
53
+ {{- '", "arguments": ' }}
54
+ {%- if tool_call.arguments is string %}
55
+ {{- tool_call.arguments }}
56
+ {%- else %}
57
+ {{- tool_call.arguments | tojson }}
58
+ {%- endif %}
59
+ {{- '}\n</tool_call>' }}
60
+ {%- endfor %}
61
+ {%- endif %}
62
+ {{- '<|im_end|>' }}
63
+ {% endgeneration %}
64
+
65
+ {#- Handle tool response messages #}
66
+ {%- elif message.role == "tool" %}
67
+ {#- Group consecutive tool responses under one user message #}
68
+ {%- if loop.first or (messages[loop.index0 - 1].role != "tool") %}
69
+ {{- '<|im_start|>user' }}
70
+ {%- endif %}
71
+ {{- '\n<tool_response>\n' }}
72
+ {{- content }}
73
+ {{- '\n</tool_response>' }}
74
+ {%- if loop.last or (messages[loop.index0 + 1].role != "tool") %}
75
+ {{- '<|im_end|>\n' }}
76
+ {%- endif %}
77
+ {%- endif %}
78
+ {%- endfor %}
79
+
80
+ {#- Add generation prompt if requested #}
81
+ {%- if add_generation_prompt %}
82
+ {{- '<|im_start|>assistant\n' }}
83
+ {%- endif %}
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1,27 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": null,
3
+ "backend": "tokenizers",
4
+ "bos_token": "<|im_start|>",
5
+ "bos_token_id": 1,
6
+ "clean_up_tokenization_spaces": false,
7
+ "eos_token": "<|im_end|>",
8
+ "eos_token_id": 2,
9
+ "is_local": false,
10
+ "legacy": false,
11
+ "local_files_only": false,
12
+ "model_input_names": [
13
+ "input_ids",
14
+ "attention_mask"
15
+ ],
16
+ "model_max_length": 4096,
17
+ "pad_token": "<|pad|>",
18
+ "pad_token_id": 49109,
19
+ "padding_side": "left",
20
+ "sp_model_kwargs": {},
21
+ "spaces_between_special_tokens": false,
22
+ "tokenizer_class": "TokenizersBackend",
23
+ "truncation_side": "right",
24
+ "unk_token": "<|unk|>",
25
+ "unk_token_id": 0,
26
+ "use_default_system_prompt": false
27
+ }