zegek commited on
Commit
773b22a
·
verified ·
1 Parent(s): d92a871

Create app.py

Browse files
Files changed (1) hide show
  1. app.py +36 -0
app.py ADDED
@@ -0,0 +1,36 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import gradio as gr
2
+ from huggingface_hub import hf_hub_download
3
+ from llama_cpp import Llama
4
+
5
+ # Download GGUF directly from Hugging Face
6
+ model_path = hf_hub_download(
7
+ repo_id="HauhauCS/Gemma4-12B-QAT-Uncensored-HauhauCS-Balanced",
8
+ filename="Gemma4-12B-QAT-Uncensored-HauhauCS-Balanced-Q4_K_M.gguf"
9
+ )
10
+
11
+ # Load model using llama.cpp engine
12
+ llm = Llama(model_path=model_path, n_ctx=2048)
13
+
14
+ def generate(prompt, max_tokens=1024, temperature=0.6, top_p=0.9):
15
+ output = llm(
16
+ f"User: {prompt}\nAssistant:",
17
+ max_tokens=int(max_tokens),
18
+ temperature=temperature,
19
+ top_p=top_p,
20
+ stop=["User:"]
21
+ )
22
+ return output["choices"][0]["text"]
23
+
24
+ # Expose Gradio API interface
25
+ demo = gr.Interface(
26
+ fn=generate,
27
+ inputs=[
28
+ gr.Textbox(label="Prompt"),
29
+ gr.Slider(64, 2048, value=1024, label="Max Tokens"),
30
+ gr.Slider(0.1, 1.0, value=0.6, label="Temperature"),
31
+ gr.Slider(0.1, 1.0, value=0.9, label="Top P")
32
+ ],
33
+ outputs="text"
34
+ )
35
+
36
+ demo.launch()