Atlas-labs commited on
Commit
c004eeb
·
verified ·
1 Parent(s): abb0cef

Upload app.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. app.py +29 -23
app.py CHANGED
@@ -1,57 +1,63 @@
1
  import gradio as gr
2
- from huggingface_hub import InferenceClient
 
3
  import os
4
  import re
 
5
 
6
  # The model to call
7
  model_id = "Atlas-labs/mini-fable-5-qwen-merged"
8
  token = os.getenv("HF_TOKEN")
9
 
10
- # Initialize the high-speed client
11
- client = InferenceClient(model=model_id, token=token)
 
 
 
 
 
 
12
 
13
- def parse_fable_response(raw_text):
14
- # Extract thought
15
  thought = ""
16
  thought_match = re.search(r'<thought>(.*?)</thought>', raw_text, re.DOTALL | re.IGNORECASE)
17
  if thought_match:
18
  thought = thought_match.group(1).strip()
19
 
20
- # Extract answer
21
  answer = raw_text
22
  if "</thought>" in raw_text.lower():
23
  answer = re.split(r'</thought>', raw_text, flags=re.IGNORECASE)[-1]
24
 
25
- # Clean up tags
26
- answer = re.sub(r'</?thought>', '', answer, flags=re.IGNORECASE).strip()
27
- answer = re.sub(r'</?response>', '', answer, flags=re.IGNORECASE).strip()
28
 
29
  if thought:
30
  return f"💡 **Reasoning:**\n> *{thought}*\n\n{answer}"
31
  return answer
32
 
 
33
  def chat(message, history):
34
- # We send the prompt and get a fast response from the HF GPU cluster
35
  prompt = f"<thought>\nAnalyzing request: {message}\n"
 
36
 
37
- try:
38
- # Using text_generation for maximum speed and control over the CoT tags
39
- response = client.text_generation(
40
- prompt + message,
41
- max_new_tokens=512,
42
- temperature=0.7,
43
  do_sample=True,
44
- return_full_text=False
45
  )
46
- return parse_fable_response(response)
47
- except Exception as e:
48
- return f"⚠️ **Inference Error:** The model is currently loading on the server. Please try again in 30 seconds. ({str(e)})"
49
 
50
  demo = gr.ChatInterface(
51
  fn=chat,
52
- title="Mini Fable 5 (Turbo)",
53
- description="High-speed reasoning engine by Atlas Labs. Powered by Hugging Face Inference API.",
54
- examples=["Explain the theory of relativity.", "How do I build a search engine?", "Why is the sky blue?"]
55
  )
56
 
57
  if __name__ == "__main__":
 
1
  import gradio as gr
2
+ from transformers import AutoModelForCausalLM, AutoTokenizer
3
+ import torch
4
  import os
5
  import re
6
+ import spaces
7
 
8
  # The model to call
9
  model_id = "Atlas-labs/mini-fable-5-qwen-merged"
10
  token = os.getenv("HF_TOKEN")
11
 
12
+ print(f"Loading tokenizer and model...")
13
+ tokenizer = AutoTokenizer.from_pretrained(model_id, token=token)
14
+ model = AutoModelForCausalLM.from_pretrained(
15
+ model_id,
16
+ torch_dtype=torch.float16,
17
+ device_map="auto",
18
+ token=token
19
+ )
20
 
21
+ def parse_fable_response(raw_text, user_input):
 
22
  thought = ""
23
  thought_match = re.search(r'<thought>(.*?)</thought>', raw_text, re.DOTALL | re.IGNORECASE)
24
  if thought_match:
25
  thought = thought_match.group(1).strip()
26
 
 
27
  answer = raw_text
28
  if "</thought>" in raw_text.lower():
29
  answer = re.split(r'</thought>', raw_text, flags=re.IGNORECASE)[-1]
30
 
31
+ answer = re.sub(r'</?thought>', '', answer, flags=re.IGNORECASE)
32
+ answer = re.sub(r'</?response>', '', answer, flags=re.IGNORECASE)
33
+ answer = answer.replace(user_input, "").strip()
34
 
35
  if thought:
36
  return f"💡 **Reasoning:**\n> *{thought}*\n\n{answer}"
37
  return answer
38
 
39
+ @spaces.GPU
40
  def chat(message, history):
 
41
  prompt = f"<thought>\nAnalyzing request: {message}\n"
42
+ inputs = tokenizer(prompt + message, return_tensors="pt").to(model.device)
43
 
44
+ with torch.no_grad():
45
+ outputs = model.generate(
46
+ **inputs,
47
+ max_new_tokens=512,
48
+ temperature=0.7,
 
49
  do_sample=True,
50
+ pad_token_id=tokenizer.eos_token_id
51
  )
52
+
53
+ raw_response = tokenizer.decode(outputs[0], skip_special_tokens=True)
54
+ return parse_fable_response(raw_response, message)
55
 
56
  demo = gr.ChatInterface(
57
  fn=chat,
58
+ title="Mini Fable 5 (ZeroGPU)",
59
+ description="High-speed reasoning engine by Atlas Labs. Powered by Hugging Face ZeroGPU.",
60
+ examples=["Explain the theory of relativity.", "Write a Python script for a binary search.", "Why is the sky blue?"]
61
  )
62
 
63
  if __name__ == "__main__":