Debugger.LLM / app.py
luguog's picture
Update app.py
e56595d verified
Raw History Blame
9.64 kB
import os, io, json, hashlib, datetime, threading
from pathlib import Path
import numpy as np
import gradio as gr
import librosa
from PIL import Image
import torch
from transformers import pipeline, AutoProcessor, WhisperForConditionalGeneration
from datasets import Dataset, Features, Value, Audio, Image as HFImage
# ---------- Runtime / device ----------
def device_map():
if torch.cuda.is_available():
return {"device": "cuda", "dtype": torch.float16}
if torch.backends.mps.is_available(): # Apple
return {"device": "mps", "dtype": torch.float16}
return {"device": "cpu", "dtype": torch.float32}
RUNTIME = device_map()
# ---------- Persistence (local ETL) ----------
DATA_DIR = Path("./data")
ART_DIR = DATA_DIR / "art"
DB_PARQUET = DATA_DIR / "fart_db.parquet"
DATA_DIR.mkdir(parents=True, exist_ok=True)
ART_DIR.mkdir(parents=True, exist_ok=True)
_db_lock = threading.Lock()
DB_FEATURES = Features({
"id": Value("string"),
"timestamp": Value("string"),
"audio_path": Value("string"),
"idea": Value("string"),
"fart_type": Value("string"),
"art_path": Value("string"),
"review": Value("string"),
"laugh_verified": Value("bool"),
"token_value": Value("float32"),
})
def load_ledger() -> Dataset:
if DB_PARQUET.exists():
return Dataset.from_parquet(str(DB_PARQUET))
return Dataset.from_dict({k: [] for k in DB_FEATURES.keys()}).cast(DB_FEATURES)
def save_ledger(ds: Dataset) -> None:
with _db_lock:
ds.to_parquet(str(DB_PARQUET))
LEDGER = load_ledger()
# ---------- Models (lazy init where expensive) ----------
# Audio classifier (emotion model repurposed as proxy demo; deterministic label selection)
_audio_cls = pipeline(
"audio-classification",
model="superb/hubert-base-superb-er",
device=0 if RUNTIME["device"] == "cuda" else -1
)
# Whisper small (local)
_processor = AutoProcessor.from_pretrained("openai/whisper-small")
_asr = WhisperForConditionalGeneration.from_pretrained("openai/whisper-small")
_asr = _asr.to(RUNTIME["device"]).to(dtype=RUNTIME["dtype"])
# Optional: Stable Diffusion (will be disabled if no GPU; safe fallback image if CPU-only)
_sd_pipe = None
if RUNTIME["device"] == "cuda":
try:
from diffusers import StableDiffusionPipeline
_sd_pipe = StableDiffusionPipeline.from_pretrained(
"stabilityai/stable-diffusion-2-1",
torch_dtype=torch.float16
).to("cuda")
except Exception:
_sd_pipe = None
# ---------- Utility ----------
def _mono_float32(wave: np.ndarray) -> np.ndarray:
if wave.ndim == 2:
wave = np.mean(wave, axis=1)
wave = wave.astype(np.float32)
# normalize if outside [-1,1]
mx = np.max(np.abs(wave)) + 1e-8
if mx > 1.0:
wave = wave / mx
return wave
def _hash_bytes(b: bytes) -> str:
return hashlib.sha256(b).hexdigest()
def _deterministic_value(s: str) -> float:
# Map SHA256 -> [0, 1) via first 8 bytes
h = hashlib.sha256(s.encode()).digest()[:8]
n = int.from_bytes(h, "big")
return (n % 10_000_000) / 10_000_000.0
# ---------- Core pipeline ----------
def detect_fart(audio_tuple):
sr, wave = audio_tuple
wave = _mono_float32(wave)
out = _audio_cls({"array": wave, "sampling_rate": sr})
# Take top label
label = max(out, key=lambda x: float(x["score"]))["label"]
return label
def transcribe_idea(audio_tuple):
sr, wave = audio_tuple
wave = _mono_float32(wave)
inputs = _processor(wave, sampling_rate=sr, return_tensors="pt")
with torch.inference_mode():
input_feats = inputs.input_features.to(RUNTIME["device"])
pred_ids = _asr.generate(input_feats)
text = _processor.batch_decode(pred_ids, skip_special_tokens=True)[0].strip()
return text
def generate_art(idea: str, art_id: str) -> str:
out_path = ART_DIR / f"{art_id}.png"
if _sd_pipe is None:
# Fallback: render text as simple image (CPU-safe)
img = Image.new("RGB", (768, 512), (0, 0, 0))
# Minimal pillow text (no additional deps): leave clean black image with no text to avoid font issues
img.save(out_path)
return str(out_path)
prompt = f"surreal, absurd, high-contrast, orange accents on black, {idea}"
with torch.inference_mode():
img = _sd_pipe(prompt).images[0]
img.save(out_path)
return str(out_path)
def review_text(idea: str, fart_type: str) -> str:
# Lightweight deterministic roast without LLM (keyless)
base = f"VC Review | type={fart_type} | idea='{idea[:120]}'"
score = _deterministic_value(idea + fart_type)
tier = "reject" if score < 0.33 else ("revise" if score < 0.66 else "fund")
return f"{base} | decision={tier} | score={score:.3f}"
def laugh_to_mint(audio_tuple) -> bool:
sr, wave = audio_tuple
wave = _mono_float32(wave)
# Energy-based laugh heuristic (deterministic threshold on log-energy variance)
frame = max(2048, int(0.05 * sr))
hop = frame // 2
rmse = librosa.feature.rms(y=wave, frame_length=frame, hop_length=hop)[0]
v = float(np.var(np.log(rmse + 1e-8)))
return v > 0.25
def create_capsule(audio_tuple, idea: str, art_path: str, review: str, fart_type: str):
sr, wave = audio_tuple
wave = _mono_float32(wave)
timestamp = datetime.datetime.utcnow().replace(tzinfo=datetime.timezone.utc).isoformat()
audio_bytes = wave.tobytes()
audio_hash = _hash_bytes(audio_bytes)
cid = f"fart-{audio_hash[:8]}"
# Persist audio as WAV (float32 PCM via soundfile; avoid PyAV)
audio_path = DATA_DIR / f"{cid}.wav"
try:
import soundfile as sf
sf.write(str(audio_path), wave, sr)
except Exception:
# Fallback: numpy npy
np.save(str(DATA_DIR / f"{cid}.npy"), wave)
audio_path = DATA_DIR / f"{cid}.npy"
value = 0.1 + 0.2 * _deterministic_value(idea) + 0.3 * _deterministic_value(review)
capsule = {
"id": cid,
"timestamp": timestamp,
"audio_path": str(audio_path),
"idea": idea,
"fart_type": fart_type,
"art_path": art_path,
"review": review,
"laugh_verified": True,
"token_value": float(value),
}
return capsule
def add_to_ledger(capsule: dict) -> Dataset:
global LEDGER
with _db_lock:
LEDGER = LEDGER.add_item(capsule)
save_ledger(LEDGER)
return LEDGER
# ---------- Gradio UI ----------
THEME_CSS = """
.gradio-container {background-color:#0b0b0b}
button, .tab-nav button {border-radius:10px}
:root {--button-primary-background-fill:#ff7a00; --button-primary-text-color:#000}
label, .markdown-body, .label-wrap, .tabs {color:#ffb26b}
"""
state_capsule = gr.State(value=None)
def process_handler(audio):
# audio: dict or tuple depending on Gradio; normalize to (sr, np.ndarray)
if isinstance(audio, dict):
sr, wave = audio["sample_rate"], np.array(audio["data"], dtype=np.float32)
else:
sr, wave = audio # already (sr, np.ndarray)
audio_tuple = (sr, wave)
fart_type = detect_fart(audio_tuple)
idea = transcribe_idea(audio_tuple)
art_id = _hash_bytes(wave.tobytes())[:8]
art_path = generate_art(idea, art_id)
review = review_text(idea, fart_type)
minted = laugh_to_mint(audio_tuple)
capsule = create_capsule(audio_tuple, idea, art_path, review, fart_type if minted else "unverified")
# Persist capsule only on Mint click; here we just stage it.
state_capsule.value = capsule
# Return UI-friendly payloads
fart_audio_value = (sr, wave) # gr.Audio expects (sr, np.ndarray)
art_img = Image.open(art_path)
review_audio_none = None # No TTS (keyless)
return fart_audio_value, idea, art_img, review, review_audio_none, json.dumps(capsule, indent=2)
def mint_handler(staged_json):
# `staged_json` is the JSON textbox value from last process; prevent mint without stage
try:
capsule = json.loads(staged_json)
except Exception:
return None, "Mint failed: no staged capsule."
ds = add_to_ledger(capsule)
# Build small table for UI
tail = ds.to_pandas().tail(10)[["id", "idea", "fart_type", "token_value"]]
return tail, f"Minted {capsule['id']}"
with gr.Blocks(title="FARTORY™ v1.0", css=THEME_CSS, theme=gr.themes.Default()) as ui:
gr.Markdown("## FART-AS-INFRASTRUCTURE™ • orange/black")
with gr.Row():
record_btn = gr.Button("💨 PROCESS MIC INPUT", variant="primary")
mint_btn = gr.Button("🪙 MINT FARTWORK", variant="secondary")
with gr.Tab("Fartifact™ Capsule"):
output_audio = gr.Audio(label="Fart Audio", interactive=False, type="numpy")
output_idea = gr.Textbox(label="Idea Capsule")
output_art = gr.Image(label="Art", type="pil")
output_review = gr.Textbox(label="VC Verdict")
tts_output = gr.Audio(label="Review Audio (off)", interactive=False, type="numpy")
staged_capsule = gr.Textbox(label="Staged Capsule (JSON)", interactive=False)
with gr.Tab("Fart Ledger"):
dataset_view = gr.Dataframe(headers=["id", "idea", "fart_type", "token_value"])
mint_status = gr.Markdown("")
mic = gr.Audio(sources=["microphone"], label="Microphone", type="numpy")
record_btn.click(
fn=process_handler,
inputs=[mic],
outputs=[output_audio, output_idea, output_art, output_review, tts_output, staged_capsule]
)
mint_btn.click(
fn=mint_handler,
inputs=[staged_capsule],
outputs=[dataset_view, mint_status]
)
if __name__ == "__main__":
ui.launch(server_port=7860, share=False)