kaoruhotarubi commited on
Commit
2ec7287
·
1 Parent(s): 5250c6c

optimized for new video

Browse files
Files changed (1) hide show
  1. app.py +6 -11
app.py CHANGED
@@ -1,25 +1,21 @@
1
  from transformers import AutoTokenizer, AutoModelForCausalLM
 
2
  import gradio as gr
3
 
4
- # Define the model name
5
- model_name = "tiiuae/Falcon3-10B-Instruct"
6
 
7
- # Load the tokenizer and model
8
  tokenizer = AutoTokenizer.from_pretrained(model_name)
9
  model = AutoModelForCausalLM.from_pretrained(
10
  model_name,
11
- torch_dtype="auto",
12
- device_map="auto"
13
  )
14
 
15
- # Define a simple chat function
16
  def chat(input_text):
17
- inputs = tokenizer(input_text, return_tensors="pt")
18
  outputs = model.generate(**inputs, max_new_tokens=100, do_sample=True)
19
- response = tokenizer.decode(outputs[0], skip_special_tokens=True)
20
- return response
21
 
22
- # Create a Gradio interface
23
  interface = gr.Interface(
24
  fn=chat,
25
  inputs=gr.Textbox(label="Your Message"),
@@ -27,5 +23,4 @@ interface = gr.Interface(
27
  title="Falcon-10B-Instruct Chat"
28
  )
29
 
30
- # Launch the app
31
  interface.launch()
 
1
  from transformers import AutoTokenizer, AutoModelForCausalLM
2
+ import torch
3
  import gradio as gr
4
 
5
+ model_name = "tiiuae/falcon3-10b-instruct"
 
6
 
 
7
  tokenizer = AutoTokenizer.from_pretrained(model_name)
8
  model = AutoModelForCausalLM.from_pretrained(
9
  model_name,
10
+ torch_dtype=torch.float16, # Use FP16 precision
11
+ device_map="auto" # Automatically maps to GPU
12
  )
13
 
 
14
  def chat(input_text):
15
+ inputs = tokenizer(input_text, return_tensors="pt").to('cuda')
16
  outputs = model.generate(**inputs, max_new_tokens=100, do_sample=True)
17
+ return tokenizer.decode(outputs[0], skip_special_tokens=True)
 
18
 
 
19
  interface = gr.Interface(
20
  fn=chat,
21
  inputs=gr.Textbox(label="Your Message"),
 
23
  title="Falcon-10B-Instruct Chat"
24
  )
25
 
 
26
  interface.launch()