import os, io, json, hashlib, datetime, threading from pathlib import Path import numpy as np import gradio as gr import librosa from PIL import Image import torch from transformers import pipeline, AutoProcessor, WhisperForConditionalGeneration from datasets import Dataset, Features, Value, Audio, Image as HFImage # ---------- Runtime / device ---------- def device_map(): if torch.cuda.is_available(): return {"device": "cuda", "dtype": torch.float16} if torch.backends.mps.is_available(): # Apple return {"device": "mps", "dtype": torch.float16} return {"device": "cpu", "dtype": torch.float32} RUNTIME = device_map() # ---------- Persistence (local ETL) ---------- DATA_DIR = Path("./data") ART_DIR = DATA_DIR / "art" DB_PARQUET = DATA_DIR / "fart_db.parquet" DATA_DIR.mkdir(parents=True, exist_ok=True) ART_DIR.mkdir(parents=True, exist_ok=True) _db_lock = threading.Lock() DB_FEATURES = Features({ "id": Value("string"), "timestamp": Value("string"), "audio_path": Value("string"), "idea": Value("string"), "fart_type": Value("string"), "art_path": Value("string"), "review": Value("string"), "laugh_verified": Value("bool"), "token_value": Value("float32"), }) def load_ledger() -> Dataset: if DB_PARQUET.exists(): return Dataset.from_parquet(str(DB_PARQUET)) return Dataset.from_dict({k: [] for k in DB_FEATURES.keys()}).cast(DB_FEATURES) def save_ledger(ds: Dataset) -> None: with _db_lock: ds.to_parquet(str(DB_PARQUET)) LEDGER = load_ledger() # ---------- Models (lazy init where expensive) ---------- # Audio classifier (emotion model repurposed as proxy demo; deterministic label selection) _audio_cls = pipeline( "audio-classification", model="superb/hubert-base-superb-er", device=0 if RUNTIME["device"] == "cuda" else -1 ) # Whisper small (local) _processor = AutoProcessor.from_pretrained("openai/whisper-small") _asr = WhisperForConditionalGeneration.from_pretrained("openai/whisper-small") _asr = _asr.to(RUNTIME["device"]).to(dtype=RUNTIME["dtype"]) # Optional: Stable Diffusion (will be disabled if no GPU; safe fallback image if CPU-only) _sd_pipe = None if RUNTIME["device"] == "cuda": try: from diffusers import StableDiffusionPipeline _sd_pipe = StableDiffusionPipeline.from_pretrained( "stabilityai/stable-diffusion-2-1", torch_dtype=torch.float16 ).to("cuda") except Exception: _sd_pipe = None # ---------- Utility ---------- def _mono_float32(wave: np.ndarray) -> np.ndarray: if wave.ndim == 2: wave = np.mean(wave, axis=1) wave = wave.astype(np.float32) # normalize if outside [-1,1] mx = np.max(np.abs(wave)) + 1e-8 if mx > 1.0: wave = wave / mx return wave def _hash_bytes(b: bytes) -> str: return hashlib.sha256(b).hexdigest() def _deterministic_value(s: str) -> float: # Map SHA256 -> [0, 1) via first 8 bytes h = hashlib.sha256(s.encode()).digest()[:8] n = int.from_bytes(h, "big") return (n % 10_000_000) / 10_000_000.0 # ---------- Core pipeline ---------- def detect_fart(audio_tuple): sr, wave = audio_tuple wave = _mono_float32(wave) out = _audio_cls({"array": wave, "sampling_rate": sr}) # Take top label label = max(out, key=lambda x: float(x["score"]))["label"] return label def transcribe_idea(audio_tuple): sr, wave = audio_tuple wave = _mono_float32(wave) inputs = _processor(wave, sampling_rate=sr, return_tensors="pt") with torch.inference_mode(): input_feats = inputs.input_features.to(RUNTIME["device"]) pred_ids = _asr.generate(input_feats) text = _processor.batch_decode(pred_ids, skip_special_tokens=True)[0].strip() return text def generate_art(idea: str, art_id: str) -> str: out_path = ART_DIR / f"{art_id}.png" if _sd_pipe is None: # Fallback: render text as simple image (CPU-safe) img = Image.new("RGB", (768, 512), (0, 0, 0)) # Minimal pillow text (no additional deps): leave clean black image with no text to avoid font issues img.save(out_path) return str(out_path) prompt = f"surreal, absurd, high-contrast, orange accents on black, {idea}" with torch.inference_mode(): img = _sd_pipe(prompt).images[0] img.save(out_path) return str(out_path) def review_text(idea: str, fart_type: str) -> str: # Lightweight deterministic roast without LLM (keyless) base = f"VC Review | type={fart_type} | idea='{idea[:120]}'" score = _deterministic_value(idea + fart_type) tier = "reject" if score < 0.33 else ("revise" if score < 0.66 else "fund") return f"{base} | decision={tier} | score={score:.3f}" def laugh_to_mint(audio_tuple) -> bool: sr, wave = audio_tuple wave = _mono_float32(wave) # Energy-based laugh heuristic (deterministic threshold on log-energy variance) frame = max(2048, int(0.05 * sr)) hop = frame // 2 rmse = librosa.feature.rms(y=wave, frame_length=frame, hop_length=hop)[0] v = float(np.var(np.log(rmse + 1e-8))) return v > 0.25 def create_capsule(audio_tuple, idea: str, art_path: str, review: str, fart_type: str): sr, wave = audio_tuple wave = _mono_float32(wave) timestamp = datetime.datetime.utcnow().replace(tzinfo=datetime.timezone.utc).isoformat() audio_bytes = wave.tobytes() audio_hash = _hash_bytes(audio_bytes) cid = f"fart-{audio_hash[:8]}" # Persist audio as WAV (float32 PCM via soundfile; avoid PyAV) audio_path = DATA_DIR / f"{cid}.wav" try: import soundfile as sf sf.write(str(audio_path), wave, sr) except Exception: # Fallback: numpy npy np.save(str(DATA_DIR / f"{cid}.npy"), wave) audio_path = DATA_DIR / f"{cid}.npy" value = 0.1 + 0.2 * _deterministic_value(idea) + 0.3 * _deterministic_value(review) capsule = { "id": cid, "timestamp": timestamp, "audio_path": str(audio_path), "idea": idea, "fart_type": fart_type, "art_path": art_path, "review": review, "laugh_verified": True, "token_value": float(value), } return capsule def add_to_ledger(capsule: dict) -> Dataset: global LEDGER with _db_lock: LEDGER = LEDGER.add_item(capsule) save_ledger(LEDGER) return LEDGER # ---------- Gradio UI ---------- THEME_CSS = """ .gradio-container {background-color:#0b0b0b} button, .tab-nav button {border-radius:10px} :root {--button-primary-background-fill:#ff7a00; --button-primary-text-color:#000} label, .markdown-body, .label-wrap, .tabs {color:#ffb26b} """ state_capsule = gr.State(value=None) def process_handler(audio): # audio: dict or tuple depending on Gradio; normalize to (sr, np.ndarray) if isinstance(audio, dict): sr, wave = audio["sample_rate"], np.array(audio["data"], dtype=np.float32) else: sr, wave = audio # already (sr, np.ndarray) audio_tuple = (sr, wave) fart_type = detect_fart(audio_tuple) idea = transcribe_idea(audio_tuple) art_id = _hash_bytes(wave.tobytes())[:8] art_path = generate_art(idea, art_id) review = review_text(idea, fart_type) minted = laugh_to_mint(audio_tuple) capsule = create_capsule(audio_tuple, idea, art_path, review, fart_type if minted else "unverified") # Persist capsule only on Mint click; here we just stage it. state_capsule.value = capsule # Return UI-friendly payloads fart_audio_value = (sr, wave) # gr.Audio expects (sr, np.ndarray) art_img = Image.open(art_path) review_audio_none = None # No TTS (keyless) return fart_audio_value, idea, art_img, review, review_audio_none, json.dumps(capsule, indent=2) def mint_handler(staged_json): # `staged_json` is the JSON textbox value from last process; prevent mint without stage try: capsule = json.loads(staged_json) except Exception: return None, "Mint failed: no staged capsule." ds = add_to_ledger(capsule) # Build small table for UI tail = ds.to_pandas().tail(10)[["id", "idea", "fart_type", "token_value"]] return tail, f"Minted {capsule['id']}" with gr.Blocks(title="FARTORY™ v1.0", css=THEME_CSS, theme=gr.themes.Default()) as ui: gr.Markdown("## FART-AS-INFRASTRUCTURE™ • orange/black") with gr.Row(): record_btn = gr.Button("💨 PROCESS MIC INPUT", variant="primary") mint_btn = gr.Button("🪙 MINT FARTWORK", variant="secondary") with gr.Tab("Fartifact™ Capsule"): output_audio = gr.Audio(label="Fart Audio", interactive=False, type="numpy") output_idea = gr.Textbox(label="Idea Capsule") output_art = gr.Image(label="Art", type="pil") output_review = gr.Textbox(label="VC Verdict") tts_output = gr.Audio(label="Review Audio (off)", interactive=False, type="numpy") staged_capsule = gr.Textbox(label="Staged Capsule (JSON)", interactive=False) with gr.Tab("Fart Ledger"): dataset_view = gr.Dataframe(headers=["id", "idea", "fart_type", "token_value"]) mint_status = gr.Markdown("") mic = gr.Audio(sources=["microphone"], label="Microphone", type="numpy") record_btn.click( fn=process_handler, inputs=[mic], outputs=[output_audio, output_idea, output_art, output_review, tts_output, staged_capsule] ) mint_btn.click( fn=mint_handler, inputs=[staged_capsule], outputs=[dataset_view, mint_status] ) if __name__ == "__main__": ui.launch(server_port=7860, share=False)