File size: 3,488 Bytes
b043d83
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
{#-
    NeoMME retrieval prompt template. Not a conversation format: NeoMME is an encoder and has no chat turns.

    `task` selects the retrieval side. A sentence-transformers v6 `MultiVectorEncoder` routes
    `task="query"` / `task="document"` here when the template declares the variable.

      task="query"     ->  <query> … N x <mask>       ColBERT-style learned query expansion. <mask> is the
                                                      masked-diffusion fill token. The exporter sets N from the
                                                      checkpoint.
      task="document"  ->  <doc> …                    Text passages. Image pages take the same <doc> prefix
                                                      followed by one image token placeholder, which the processor
                                                      expands after it knows the patch grid.
-#}
{%- if task is not defined -%}
    {{- raise_exception("NeoMME chat templates require task='query' or task='document'.") -}}
{%- endif -%}
{%- if task not in ['query', 'document'] -%}
    {{- raise_exception("task=" ~ task ~ " is not supported: expected 'query' or 'document'.") -}}
{%- endif -%}
{%- if messages is not defined or not messages -%}
    {{- raise_exception("NeoMME chat conversations must contain at least one message.") -}}
{%- endif -%}

{# Collect content and validate one retrieval item without imposing batch-level policy. #}
{%- set state = namespace(text='', has_text=false, image_count=0) -%}
{%- for message in messages -%}
    {%- set content = message.content -%}
    {%- set items = [{'type': 'text', 'text': content}] if content is string else content -%}
    {%- for item in items -%}
        {%- if item.type == 'text' -%}
            {%- if image_token in item.text -%}
                {{- raise_exception(image_token ~ " is reserved for image documents.") -}}
            {%- endif -%}
            {%- set state.has_text = true -%}
            {%- set state.text = state.text + item.text -%}
        {%- elif item.type == 'image' -%}
            {%- if item.image is not defined or item.image is none or item.image == '' -%}
                {{- raise_exception("NeoMME image content must provide an image source.") -}}
            {%- endif -%}
            {%- set state.image_count = state.image_count + 1 -%}
        {%- elif item.type == 'image_url' -%}
            {%- if item.image_url is not defined or not item.image_url -%}
                {{- raise_exception("NeoMME image_url content must provide an image source.") -}}
            {%- endif -%}
            {%- set state.image_count = state.image_count + 1 -%}
        {%- else -%}
            {{- raise_exception("NeoMME chat templates do not support content type " ~ item.type ~ ".") -}}
        {%- endif -%}
    {%- endfor -%}
{%- endfor -%}

{%- if state.image_count and state.has_text -%}
    {{- raise_exception("NeoMME cannot encode text and images in the same conversation.") -}}
{%- endif -%}
{%- if state.image_count > 1 -%}
    {{- raise_exception("NeoMME accepts one image document per conversation.") -}}
{%- endif -%}
{%- if state.image_count and task != 'document' -%}
    {{- raise_exception("NeoMME image content must use task='document'.") -}}
{%- endif -%}

{%- set content = image_token if state.image_count else state.text -%}
{%- if task == 'query' -%}
    {{- query_token + content + mask_token * 10 -}}
{%- else -%}
    {{- document_token + content -}}
{%- endif -%}