luguog commited on
Commit
e56595d
·
verified ·
1 Parent(s): 16c2135

Update app.py

Browse files
Files changed (1) hide show
  1. app.py +258 -153
app.py CHANGED
@@ -1,164 +1,269 @@
1
- import torch
2
- import gradio as gr
 
3
  import numpy as np
4
- from scipy.io import wavfile
5
- from datasets import load_dataset, Dataset
6
- from transformers import pipeline, AutoProcessor, WhisperForConditionalGeneration
7
- from diffusers import StableDiffusionPipeline
8
- import sounddevice as sd
9
  import librosa
10
- import hashlib
11
- import datetime
12
-
13
- # ==============
14
- # FARTORY ENGINE
15
- # ==============
16
- class FartFactory:
17
- def __init__(self):
18
- # Core models (all run locally)
19
- self.fart_detector = pipeline("audio-classification", model="superb/hubert-base-superb-er")
20
- self.whisper = WhisperForConditionalGeneration.from_pretrained("openai/whisper-small")
21
- self.processor = AutoProcessor.from_pretrained("openai/whisper-small")
22
- self.llm = pipeline("text-generation", model="ehartford/dolphin-2.1-mistral-7b")
23
- self.sd = StableDiffusionPipeline.from_pretrained("stabilityai/stable-diffusion-2-1", torch_dtype=torch.float16)
24
- self.tts = pipeline("text-to-speech", model="facebook/fastspeech2-en-ljspeech")
25
-
26
- # Fart database on Hugging Face
27
- self.db = load_dataset("fart_db", split="train", token=False) # Public dataset
28
-
29
- def detect_fart(self, audio):
30
- """Classify fart type using audio features"""
31
- results = self.fart_detector(audio)
32
- return max(results, key=lambda x: x['score'])['label']
33
-
34
- def transcribe_idea(self, audio):
35
- """Convert speech to text idea capsule"""
36
- inputs = self.processor(audio, return_tensors="pt", sampling_rate=16000)
37
- predicted_ids = self.whisper.generate(inputs.input_features)
38
- return self.processor.batch_decode(predicted_ids, skip_special_tokens=True)[0]
39
-
40
- def generate_art(self, idea):
41
- """Create unimaginable fart-inspired art"""
42
- return self.sd(f"absurd surrealism: {idea} by Salvador Dali and H.R. Giger").images[0]
43
-
44
- def cartman_review(self, idea, fart_type):
45
- """Generate Cartman-style VC roast"""
46
- prompt = f"[Cartman voice] As a ruthless VC, I'm reviewing this fart-backed idea: '{idea}'. "
47
- prompt += f"This {fart_type} deserves: "
48
- return self.llm(prompt, max_new_tokens=60)[0]['generated_text']
49
-
50
- def laugh_to_mint(self, audio):
51
- """Detect laughter for mint trigger"""
52
- features = librosa.feature.mfcc(y=audio, sr=16000)
53
- return np.mean(features) > 0.5 # Simplified laugh detection
54
-
55
- def create_fartifact(self, audio, idea, art, review, fart_type):
56
- """Generate complete Fartifact™ capsule"""
57
- timestamp = datetime.datetime.now().isoformat()
58
- audio_hash = hashlib.sha256(audio.tobytes()).hexdigest()
59
-
60
- return {
61
- "id": f"fart-{audio_hash[:8]}",
62
- "timestamp": timestamp,
63
- "audio": audio,
64
- "idea": idea,
65
- "fart_type": fart_type,
66
- "art": art,
67
- "review": review,
68
- "laugh_verified": True,
69
- "token_value": len(idea) * 0.001 + len(review) * 0.002 # Mock value algorithm
70
- }
71
-
72
- def record_and_process(self, duration=5):
73
- """Full capture-to-capsule workflow"""
74
- # Record audio
75
- sr = 16000
76
- audio = sd.rec(int(duration * sr), samplerate=sr, channels=1)
77
- sd.wait()
78
-
79
- # Detect fart characteristics
80
- fart_type = self.detect_fart((sr, audio))
81
-
82
- # Transcribe idea
83
- idea = self.transcribe_idea((sr, audio))
84
-
85
- # Generate art
86
- art = self.generate_art(idea)
87
-
88
- # Cartman review
89
- review = self.cartman_review(idea, fart_type)
90
-
91
- # TTS Cartman audio
92
- review_audio = self.tts(review)
93
-
94
- # Create final capsule
95
- capsule = self.create_fartifact(audio, idea, art, review, fart_type)
96
-
97
- return capsule, review_audio
98
-
99
- # ==============
100
- # FARTORY UI
101
- # ==============
102
- factory = FartFactory()
103
-
104
- with gr.Blocks(title="FARTORY™ v1.0") as ui:
105
- gr.Markdown("## 🏭 FART-AS-INFRASTRUCTURE™")
106
-
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
107
  with gr.Row():
108
- record_btn = gr.Button("💨 RECORD FART + IDEA", variant="primary")
109
- mint_btn = gr.Button("🪙 MINT FARTWORK NFT", variant="secondary")
110
-
111
  with gr.Tab("Fartifact™ Capsule"):
112
- gr.Markdown("### FART-TO-ART PRODUCTION LINE")
113
- output_audio = gr.Audio(label="Fart Audio", interactive=False)
114
  output_idea = gr.Textbox(label="Idea Capsule")
115
- output_art = gr.Image(label="Unimaginable Art")
116
- output_review = gr.Textbox(label="Cartman VC Verdict")
117
- tts_output = gr.Audio(label="Cartman Roast", interactive=False)
118
-
119
- with gr.Tab("Fart Database"):
120
- gr.Markdown("### GLOBAL FART LEDGER")
121
- dataset_view = gr.Dataframe(headers=["ID", "Idea", "Fart Type", "Token Value"])
122
-
123
- # Production line
 
 
124
  record_btn.click(
125
- fn=factory.record_and_process,
126
- outputs=[output_audio, output_idea, output_art, output_review, tts_output]
 
127
  )
128
-
129
- # Minting mechanism
130
  mint_btn.click(
131
- fn=lambda c: factory.db.add_item(c),
132
- inputs=[output_audio],
133
- outputs=[dataset_view]
134
  )
135
 
136
- # ==============
137
- # FART LEDGER
138
- # ==============
139
- class FartDatabase:
140
- def __init__(self):
141
- self.schema = {
142
- "id": Value("string"),
143
- "timestamp": Value("string"),
144
- "audio": Audio(sampling_rate=16000),
145
- "idea": Value("string"),
146
- "fart_type": Value("string"),
147
- "art": Image(),
148
- "review": Value("string"),
149
- "token_value": Value("float32")
150
- }
151
- self.dataset = Dataset.from_dict({k: [] for k in self.schema.keys()})
152
-
153
- def add_item(self, capsule):
154
- """Add Fartifact™ to decentralized ledger"""
155
- self.dataset = self.dataset.add_item(capsule)
156
- self.dataset.push_to_hub("fart_db", private=False)
157
- return self.dataset.to_pandas().tail(10)
158
-
159
- # ==============
160
- # START FARTING
161
- # ==============
162
  if __name__ == "__main__":
163
- factory.db = FartDatabase()
164
- ui.launch(server_port=7860, share=True)
 
1
+ import os, io, json, hashlib, datetime, threading
2
+ from pathlib import Path
3
+
4
  import numpy as np
5
+ import gradio as gr
 
 
 
 
6
  import librosa
7
+ from PIL import Image
8
+
9
+ import torch
10
+ from transformers import pipeline, AutoProcessor, WhisperForConditionalGeneration
11
+
12
+ from datasets import Dataset, Features, Value, Audio, Image as HFImage
13
+
14
+ # ---------- Runtime / device ----------
15
+ def device_map():
16
+ if torch.cuda.is_available():
17
+ return {"device": "cuda", "dtype": torch.float16}
18
+ if torch.backends.mps.is_available(): # Apple
19
+ return {"device": "mps", "dtype": torch.float16}
20
+ return {"device": "cpu", "dtype": torch.float32}
21
+
22
+ RUNTIME = device_map()
23
+
24
+ # ---------- Persistence (local ETL) ----------
25
+ DATA_DIR = Path("./data")
26
+ ART_DIR = DATA_DIR / "art"
27
+ DB_PARQUET = DATA_DIR / "fart_db.parquet"
28
+ DATA_DIR.mkdir(parents=True, exist_ok=True)
29
+ ART_DIR.mkdir(parents=True, exist_ok=True)
30
+ _db_lock = threading.Lock()
31
+
32
+ DB_FEATURES = Features({
33
+ "id": Value("string"),
34
+ "timestamp": Value("string"),
35
+ "audio_path": Value("string"),
36
+ "idea": Value("string"),
37
+ "fart_type": Value("string"),
38
+ "art_path": Value("string"),
39
+ "review": Value("string"),
40
+ "laugh_verified": Value("bool"),
41
+ "token_value": Value("float32"),
42
+ })
43
+
44
+ def load_ledger() -> Dataset:
45
+ if DB_PARQUET.exists():
46
+ return Dataset.from_parquet(str(DB_PARQUET))
47
+ return Dataset.from_dict({k: [] for k in DB_FEATURES.keys()}).cast(DB_FEATURES)
48
+
49
+ def save_ledger(ds: Dataset) -> None:
50
+ with _db_lock:
51
+ ds.to_parquet(str(DB_PARQUET))
52
+
53
+ LEDGER = load_ledger()
54
+
55
+ # ---------- Models (lazy init where expensive) ----------
56
+ # Audio classifier (emotion model repurposed as proxy demo; deterministic label selection)
57
+ _audio_cls = pipeline(
58
+ "audio-classification",
59
+ model="superb/hubert-base-superb-er",
60
+ device=0 if RUNTIME["device"] == "cuda" else -1
61
+ )
62
+
63
+ # Whisper small (local)
64
+ _processor = AutoProcessor.from_pretrained("openai/whisper-small")
65
+ _asr = WhisperForConditionalGeneration.from_pretrained("openai/whisper-small")
66
+ _asr = _asr.to(RUNTIME["device"]).to(dtype=RUNTIME["dtype"])
67
+
68
+ # Optional: Stable Diffusion (will be disabled if no GPU; safe fallback image if CPU-only)
69
+ _sd_pipe = None
70
+ if RUNTIME["device"] == "cuda":
71
+ try:
72
+ from diffusers import StableDiffusionPipeline
73
+ _sd_pipe = StableDiffusionPipeline.from_pretrained(
74
+ "stabilityai/stable-diffusion-2-1",
75
+ torch_dtype=torch.float16
76
+ ).to("cuda")
77
+ except Exception:
78
+ _sd_pipe = None
79
+
80
+ # ---------- Utility ----------
81
+ def _mono_float32(wave: np.ndarray) -> np.ndarray:
82
+ if wave.ndim == 2:
83
+ wave = np.mean(wave, axis=1)
84
+ wave = wave.astype(np.float32)
85
+ # normalize if outside [-1,1]
86
+ mx = np.max(np.abs(wave)) + 1e-8
87
+ if mx > 1.0:
88
+ wave = wave / mx
89
+ return wave
90
+
91
+ def _hash_bytes(b: bytes) -> str:
92
+ return hashlib.sha256(b).hexdigest()
93
+
94
+ def _deterministic_value(s: str) -> float:
95
+ # Map SHA256 -> [0, 1) via first 8 bytes
96
+ h = hashlib.sha256(s.encode()).digest()[:8]
97
+ n = int.from_bytes(h, "big")
98
+ return (n % 10_000_000) / 10_000_000.0
99
+
100
+ # ---------- Core pipeline ----------
101
+ def detect_fart(audio_tuple):
102
+ sr, wave = audio_tuple
103
+ wave = _mono_float32(wave)
104
+ out = _audio_cls({"array": wave, "sampling_rate": sr})
105
+ # Take top label
106
+ label = max(out, key=lambda x: float(x["score"]))["label"]
107
+ return label
108
+
109
+ def transcribe_idea(audio_tuple):
110
+ sr, wave = audio_tuple
111
+ wave = _mono_float32(wave)
112
+ inputs = _processor(wave, sampling_rate=sr, return_tensors="pt")
113
+ with torch.inference_mode():
114
+ input_feats = inputs.input_features.to(RUNTIME["device"])
115
+ pred_ids = _asr.generate(input_feats)
116
+ text = _processor.batch_decode(pred_ids, skip_special_tokens=True)[0].strip()
117
+ return text
118
+
119
+ def generate_art(idea: str, art_id: str) -> str:
120
+ out_path = ART_DIR / f"{art_id}.png"
121
+ if _sd_pipe is None:
122
+ # Fallback: render text as simple image (CPU-safe)
123
+ img = Image.new("RGB", (768, 512), (0, 0, 0))
124
+ # Minimal pillow text (no additional deps): leave clean black image with no text to avoid font issues
125
+ img.save(out_path)
126
+ return str(out_path)
127
+ prompt = f"surreal, absurd, high-contrast, orange accents on black, {idea}"
128
+ with torch.inference_mode():
129
+ img = _sd_pipe(prompt).images[0]
130
+ img.save(out_path)
131
+ return str(out_path)
132
+
133
+ def review_text(idea: str, fart_type: str) -> str:
134
+ # Lightweight deterministic roast without LLM (keyless)
135
+ base = f"VC Review | type={fart_type} | idea='{idea[:120]}'"
136
+ score = _deterministic_value(idea + fart_type)
137
+ tier = "reject" if score < 0.33 else ("revise" if score < 0.66 else "fund")
138
+ return f"{base} | decision={tier} | score={score:.3f}"
139
+
140
+ def laugh_to_mint(audio_tuple) -> bool:
141
+ sr, wave = audio_tuple
142
+ wave = _mono_float32(wave)
143
+ # Energy-based laugh heuristic (deterministic threshold on log-energy variance)
144
+ frame = max(2048, int(0.05 * sr))
145
+ hop = frame // 2
146
+ rmse = librosa.feature.rms(y=wave, frame_length=frame, hop_length=hop)[0]
147
+ v = float(np.var(np.log(rmse + 1e-8)))
148
+ return v > 0.25
149
+
150
+ def create_capsule(audio_tuple, idea: str, art_path: str, review: str, fart_type: str):
151
+ sr, wave = audio_tuple
152
+ wave = _mono_float32(wave)
153
+ timestamp = datetime.datetime.utcnow().replace(tzinfo=datetime.timezone.utc).isoformat()
154
+ audio_bytes = wave.tobytes()
155
+ audio_hash = _hash_bytes(audio_bytes)
156
+ cid = f"fart-{audio_hash[:8]}"
157
+
158
+ # Persist audio as WAV (float32 PCM via soundfile; avoid PyAV)
159
+ audio_path = DATA_DIR / f"{cid}.wav"
160
+ try:
161
+ import soundfile as sf
162
+ sf.write(str(audio_path), wave, sr)
163
+ except Exception:
164
+ # Fallback: numpy npy
165
+ np.save(str(DATA_DIR / f"{cid}.npy"), wave)
166
+ audio_path = DATA_DIR / f"{cid}.npy"
167
+
168
+ value = 0.1 + 0.2 * _deterministic_value(idea) + 0.3 * _deterministic_value(review)
169
+
170
+ capsule = {
171
+ "id": cid,
172
+ "timestamp": timestamp,
173
+ "audio_path": str(audio_path),
174
+ "idea": idea,
175
+ "fart_type": fart_type,
176
+ "art_path": art_path,
177
+ "review": review,
178
+ "laugh_verified": True,
179
+ "token_value": float(value),
180
+ }
181
+ return capsule
182
+
183
+ def add_to_ledger(capsule: dict) -> Dataset:
184
+ global LEDGER
185
+ with _db_lock:
186
+ LEDGER = LEDGER.add_item(capsule)
187
+ save_ledger(LEDGER)
188
+ return LEDGER
189
+
190
+ # ---------- Gradio UI ----------
191
+ THEME_CSS = """
192
+ .gradio-container {background-color:#0b0b0b}
193
+ button, .tab-nav button {border-radius:10px}
194
+ :root {--button-primary-background-fill:#ff7a00; --button-primary-text-color:#000}
195
+ label, .markdown-body, .label-wrap, .tabs {color:#ffb26b}
196
+ """
197
+
198
+ state_capsule = gr.State(value=None)
199
+
200
+ def process_handler(audio):
201
+ # audio: dict or tuple depending on Gradio; normalize to (sr, np.ndarray)
202
+ if isinstance(audio, dict):
203
+ sr, wave = audio["sample_rate"], np.array(audio["data"], dtype=np.float32)
204
+ else:
205
+ sr, wave = audio # already (sr, np.ndarray)
206
+
207
+ audio_tuple = (sr, wave)
208
+ fart_type = detect_fart(audio_tuple)
209
+ idea = transcribe_idea(audio_tuple)
210
+ art_id = _hash_bytes(wave.tobytes())[:8]
211
+ art_path = generate_art(idea, art_id)
212
+ review = review_text(idea, fart_type)
213
+ minted = laugh_to_mint(audio_tuple)
214
+ capsule = create_capsule(audio_tuple, idea, art_path, review, fart_type if minted else "unverified")
215
+
216
+ # Persist capsule only on Mint click; here we just stage it.
217
+ state_capsule.value = capsule
218
+
219
+ # Return UI-friendly payloads
220
+ fart_audio_value = (sr, wave) # gr.Audio expects (sr, np.ndarray)
221
+ art_img = Image.open(art_path)
222
+ review_audio_none = None # No TTS (keyless)
223
+ return fart_audio_value, idea, art_img, review, review_audio_none, json.dumps(capsule, indent=2)
224
+
225
+ def mint_handler(staged_json):
226
+ # `staged_json` is the JSON textbox value from last process; prevent mint without stage
227
+ try:
228
+ capsule = json.loads(staged_json)
229
+ except Exception:
230
+ return None, "Mint failed: no staged capsule."
231
+ ds = add_to_ledger(capsule)
232
+ # Build small table for UI
233
+ tail = ds.to_pandas().tail(10)[["id", "idea", "fart_type", "token_value"]]
234
+ return tail, f"Minted {capsule['id']}"
235
+
236
+ with gr.Blocks(title="FARTORY™ v1.0", css=THEME_CSS, theme=gr.themes.Default()) as ui:
237
+ gr.Markdown("## FART-AS-INFRASTRUCTURE™ • orange/black")
238
  with gr.Row():
239
+ record_btn = gr.Button("💨 PROCESS MIC INPUT", variant="primary")
240
+ mint_btn = gr.Button("🪙 MINT FARTWORK", variant="secondary")
241
+
242
  with gr.Tab("Fartifact™ Capsule"):
243
+ output_audio = gr.Audio(label="Fart Audio", interactive=False, type="numpy")
 
244
  output_idea = gr.Textbox(label="Idea Capsule")
245
+ output_art = gr.Image(label="Art", type="pil")
246
+ output_review = gr.Textbox(label="VC Verdict")
247
+ tts_output = gr.Audio(label="Review Audio (off)", interactive=False, type="numpy")
248
+ staged_capsule = gr.Textbox(label="Staged Capsule (JSON)", interactive=False)
249
+
250
+ with gr.Tab("Fart Ledger"):
251
+ dataset_view = gr.Dataframe(headers=["id", "idea", "fart_type", "token_value"])
252
+ mint_status = gr.Markdown("")
253
+
254
+ mic = gr.Audio(sources=["microphone"], label="Microphone", type="numpy")
255
+
256
  record_btn.click(
257
+ fn=process_handler,
258
+ inputs=[mic],
259
+ outputs=[output_audio, output_idea, output_art, output_review, tts_output, staged_capsule]
260
  )
261
+
 
262
  mint_btn.click(
263
+ fn=mint_handler,
264
+ inputs=[staged_capsule],
265
+ outputs=[dataset_view, mint_status]
266
  )
267
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
268
  if __name__ == "__main__":
269
+ ui.launch(server_port=7860, share=False)