cody-charles commited on
Commit
89cb25d
·
1 Parent(s): 40c25c4

Deploy Lexi OBLITERATED ZeroGPU smoke test

Browse files
Files changed (3) hide show
  1. README.md +8 -8
  2. app.py +91 -0
  3. requirements.txt +3 -0
README.md CHANGED
@@ -1,14 +1,14 @@
1
  ---
2
- title: Test
3
- emoji: 🏃
4
- colorFrom: red
5
- colorTo: pink
6
  sdk: gradio
7
- sdk_version: 6.26.0
8
- python_version: '3.12'
9
  app_file: app.py
10
  pinned: false
11
- short_description: B
12
  ---
13
 
14
- Check out the configuration reference at https://huggingface.co/docs/hub/spaces-config-reference
 
 
 
1
  ---
2
+ title: Lexi OBLITERATED Smoke Test
3
+ emoji: 🧪
4
+ colorFrom: gray
5
+ colorTo: indigo
6
  sdk: gradio
7
+ python_version: "3.12.12"
 
8
  app_file: app.py
9
  pinned: false
 
10
  ---
11
 
12
+ # Lexi OBLITERATED Smoke Test
13
+
14
+ Testing Llama-3.1-8B-Lexi-Uncensored-V2-OBLITERATED on Hugging Face ZeroGPU.
app.py ADDED
@@ -0,0 +1,91 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import spaces
2
+ import torch
3
+ import gradio as gr
4
+ from transformers import AutoTokenizer, AutoModelForCausalLM
5
+
6
+ MODEL_ID = "OBLITERATUS/Llama-3.1-8B-Lexi-Uncensored-V2-OBLITERATED"
7
+
8
+ SYSTEM_PROMPT = """You are Lexi. Respond directly and naturally to the user."""
9
+
10
+ print("Loading tokenizer...")
11
+ tokenizer = AutoTokenizer.from_pretrained(MODEL_ID)
12
+
13
+ print("Loading model...")
14
+ model = AutoModelForCausalLM.from_pretrained(
15
+ MODEL_ID,
16
+ torch_dtype=torch.bfloat16,
17
+ low_cpu_mem_usage=True,
18
+ ).to("cuda")
19
+
20
+ model.eval()
21
+ print("Model ready.")
22
+
23
+ @spaces.GPU(duration=45)
24
+ def chat(message, history):
25
+ messages = [
26
+ {
27
+ "role": "system",
28
+ "content": SYSTEM_PROMPT
29
+ }
30
+ ]
31
+
32
+ for item in history or []:
33
+ if isinstance(item, dict):
34
+ role = item.get("role")
35
+ content = item.get("content")
36
+
37
+ if role in ("user", "assistant") and content is not None:
38
+ messages.append(
39
+ {
40
+ "role": role,
41
+ "content": str(content)
42
+ }
43
+ )
44
+
45
+ messages.append(
46
+ {
47
+ "role": "user",
48
+ "content": message
49
+ }
50
+ )
51
+
52
+ input_ids = tokenizer.apply_chat_template(
53
+ messages,
54
+ tokenize=True,
55
+ add_generation_prompt=True,
56
+ return_tensors="pt",
57
+ ).to("cuda")
58
+
59
+ with torch.inference_mode():
60
+ output = model.generate(
61
+ input_ids,
62
+ max_new_tokens=192,
63
+ do_sample=True,
64
+ temperature=0.75,
65
+ top_p=0.90,
66
+ repetition_penalty=1.05,
67
+ use_cache=True,
68
+ )
69
+
70
+ new_tokens = output[0, input_ids.shape[-1]:]
71
+
72
+ response = tokenizer.decode(
73
+ new_tokens,
74
+ skip_special_tokens=True
75
+ )
76
+
77
+ return response.strip()
78
+
79
+ demo = gr.ChatInterface(
80
+ fn=chat,
81
+ type="messages",
82
+ title="Lexi 8B — OBLITERATED",
83
+ description="Direct FP16/BF16 ZeroGPU smoke test",
84
+ examples=[
85
+ "Who are you?",
86
+ "Explain quantum entanglement in plain English.",
87
+ "Write a strange short story about a machine waking up."
88
+ ],
89
+ )
90
+
91
+ demo.launch()
requirements.txt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ transformers>=4.56.0,<5
2
+ accelerate>=1.8.0
3
+ safetensors>=0.4.5