mohit-sarvam commited on
Commit
a198288
·
verified ·
1 Parent(s): 729f7b8

Update README.md

Browse files
Files changed (1) hide show
  1. README.md +41 -33
README.md CHANGED
@@ -103,50 +103,46 @@ base_url = "https://api.sarvam.ai/v1"
103
  model_name = "sarvam-m"
104
  api_key = "Your-API-Key" # get it from https://dashboard.sarvam.ai/
105
 
106
-
107
  client = OpenAI(
108
  base_url=base_url,
109
  api_key=api_key,
110
  ).with_options(max_retries=1)
111
 
112
- response = client.chat.completions.create(
 
 
 
 
 
113
  model=model_name,
114
- messages=[
115
- {"role": "system", "content": "say hi"},
116
- {"role": "user", "content": "say hi"},
117
- ],
118
- stream=False,
119
- max_completion_tokens=2048,
120
- # reasoning_effort="low", # set either of 3 values to enable reasoning
121
- )
122
- print(response.choices[0].message.content)
123
-
124
- response1 = client.chat.completions.create(
125
- model=model_name,
126
- messages=[
127
- {"role": "system", "content": "You're a helpful AI assistant"},
128
- {"role": "user", "content": "Explain quantum computing in simple terms"}
129
- ],
130
  max_completion_tokens=4096,
131
- reasoning_effort="medium" # Optional reasoning mode
132
- )
133
- print("First response:", response1.choices[0].message.content)
134
-
135
- # Second turn (using previous response as context)
136
- response2 = client.chat.completions.create(
137
- model=model_name,
138
- messages=[
139
- {"role": "system", "content": "You're a helpful AI assistant"},
140
- {"role": "user", "content": "Explain quantum computing in simple terms"},
141
- {"role": "assistant", "content": response1.choices[0].message.content}, # Previous response
142
- {"role": "user", "content": "Can you give an analogy for superposition?"}
143
- ],
 
 
 
 
144
  reasoning_effort="high",
145
  max_completion_tokens=8192,
146
- )
147
  print("Follow-up response:", response2.choices[0].message.content)
148
  ```
149
 
 
 
150
  # VLLM Deployment
151
 
152
  For easy deployment, we can use `vllm>=0.8.5` and create an OpenAI-compatible API endpoint with `vllm serve sarvamai/sarvam-m`
@@ -190,4 +186,16 @@ print("content:", content)
190
  messages.append(
191
  {"role": "assistant", "content": output_text}
192
  )
193
- ```
 
 
 
 
 
 
 
 
 
 
 
 
 
103
  model_name = "sarvam-m"
104
  api_key = "Your-API-Key" # get it from https://dashboard.sarvam.ai/
105
 
 
106
  client = OpenAI(
107
  base_url=base_url,
108
  api_key=api_key,
109
  ).with_options(max_retries=1)
110
 
111
+ messages = [
112
+ {"role": "system", "content": "You're a helpful AI assistant"},
113
+ {"role": "user", "content": "Explain quantum computing in simple terms"},
114
+ ]
115
+
116
+ response1 = client.chat.completions.create(
117
  model=model_name,
118
+ messages=messages,
119
+ reasoning_effort="medium", # Optional reasoning mode
 
 
 
 
 
 
 
 
 
 
 
 
 
 
120
  max_completion_tokens=4096,
121
+ )
122
+ print("First response:", response1.choices[0].message.content)
123
+
124
+ # Building messages for the second turn (using previous response as context)
125
+ messages.extend(
126
+ [
127
+ {
128
+ "role": "assistant",
129
+ "content": response1.choices[0].message.content,
130
+ },
131
+ {"role": "user", "content": "Can you give an analogy for superposition?"},
132
+ ]
133
+ )
134
+
135
+ response2 = client.chat.completions.create(
136
+ model=model_name,
137
+ messages=messages,
138
  reasoning_effort="high",
139
  max_completion_tokens=8192,
140
+ )
141
  print("Follow-up response:", response2.choices[0].message.content)
142
  ```
143
 
144
+ Refer to API docs here: [sarvam API docs](https://docs.sarvam.ai/api-reference-docs/introduction)
145
+
146
  # VLLM Deployment
147
 
148
  For easy deployment, we can use `vllm>=0.8.5` and create an OpenAI-compatible API endpoint with `vllm serve sarvamai/sarvam-m`
 
186
  messages.append(
187
  {"role": "assistant", "content": output_text}
188
  )
189
+ ```
190
+
191
+ # Running the model on a CPU
192
+
193
+ The repo contains bf16 and q8 gguf files built using https://github.com/ggml-org/llama.cpp/blob/master/docs/build.md#cpu-build
194
+
195
+ You can use the model using cli as explained in docs https://github.com/ggml-org/llama.cpp/tree/master/tools/main
196
+
197
+ Example Command:
198
+
199
+ `./build/bin/llama-cli -i -m /projects/data/romit_sarvam_ai/models/gguf/sarvam-m-q8_0.gguf -c 8192 -t 16`
200
+
201
+ We got about 4 tokens per second on 16 cores.