# `spaces` MUST be imported before torch on Hugging Face ZeroGPU. # Hugging Face patches CUDA when this package is imported. import spaces # Standard-library imports. from pathlib import Path import json # Gradio builds the Hugging Face Space UI. import gradio as gr # PyTorch runs the actual WVY model. # This comes AFTER `spaces` intentionally. import torch # This is the exact tokenizer library used by the training notebook. from tokenizers import Tokenizer # These are the model classes from the model.py that came with the Space. from model import LiquidModelConfig, TinyLiquidCausalLanguageModel # ============================================================ # FILE LOCATIONS # ============================================================ # The Space keeps app.py, model.py, tokenizer.json, # model_config.json, and the checkpoint in the same directory. ROOT = Path(__file__).resolve().parent TOKENIZER_PATH = ROOT / "tokenizer.json" CONFIG_PATH = ROOT / "model_config.json" CHECKPOINT_PATH = ROOT / "tiny_liquid_causal_lm.pt" # ============================================================ # ORIGINAL MODEL SETTINGS # ============================================================ # This is copied from the actual training notebook. # # SEQUENCE_LENGTH = 256 # # The model was trained on 256-token chunks, so generation # gives the model at most the latest 256 tokens every step. SEQUENCE_LENGTH = 256 # ============================================================ # ALL SPECIAL TOKENS # ============================================================ # These are the EXACT seven special tokens used when the # tokenizer was created in the notebook. PAD_TOKEN = "<|pad|>" UNK_TOKEN = "<|unk|>" BOS_TOKEN = "<|bos|>" EOS_TOKEN = "<|eos|>" USER_TOKEN = "<|user|>" ASSISTANT_TOKEN = "<|assistant|>" SYSTEM_TOKEN = "<|system|>" # This preserves the exact ordering used during tokenizer training. # # In the tokenizer you trained, this corresponds to: # # <|pad|> = 0 # <|unk|> = 1 # <|bos|> = 2 # <|eos|> = 3 # <|user|> = 4 # <|assistant|> = 5 # <|system|> = 6 SPECIAL_TOKENS = [ PAD_TOKEN, UNK_TOKEN, BOS_TOKEN, EOS_TOKEN, USER_TOKEN, ASSISTANT_TOKEN, SYSTEM_TOKEN, ] # ============================================================ # LOAD TOKENIZER # ============================================================ # Load the exact tokenizer.json produced by the training notebook. # We do NOT recreate or retrain a tokenizer inside the Space. tokenizer = Tokenizer.from_file(str(TOKENIZER_PATH)) # Read every special-token ID directly from tokenizer.json. # This avoids manually assuming IDs in the actual runtime code. PAD_ID = tokenizer.token_to_id(PAD_TOKEN) UNK_ID = tokenizer.token_to_id(UNK_TOKEN) BOS_ID = tokenizer.token_to_id(BOS_TOKEN) EOS_ID = tokenizer.token_to_id(EOS_TOKEN) USER_ID = tokenizer.token_to_id(USER_TOKEN) ASSISTANT_ID = tokenizer.token_to_id(ASSISTANT_TOKEN) SYSTEM_ID = tokenizer.token_to_id(SYSTEM_TOKEN) # Keep all of the token mappings together so startup can verify them. SPECIAL_TOKEN_IDS = { PAD_TOKEN: PAD_ID, UNK_TOKEN: UNK_ID, BOS_TOKEN: BOS_ID, EOS_TOKEN: EOS_ID, USER_TOKEN: USER_ID, ASSISTANT_TOKEN: ASSISTANT_ID, SYSTEM_TOKEN: SYSTEM_ID, } # Fail immediately if the wrong tokenizer gets uploaded to the Space. # Every one of these tokens existed during training. for token_name, token_id in SPECIAL_TOKEN_IDS.items(): if token_id is None: raise RuntimeError( f"Tokenizer is missing required special token: {token_name}" ) # The uploaded tokenizer from the model bundle has these exact IDs. # Verifying them catches an accidental tokenizer replacement. EXPECTED_SPECIAL_TOKEN_IDS = { PAD_TOKEN: 0, UNK_TOKEN: 1, BOS_TOKEN: 2, EOS_TOKEN: 3, USER_TOKEN: 4, ASSISTANT_TOKEN: 5, SYSTEM_TOKEN: 6, } for token_name, expected_id in EXPECTED_SPECIAL_TOKEN_IDS.items(): actual_id = SPECIAL_TOKEN_IDS[token_name] if actual_id != expected_id: raise RuntimeError( f"Special token mismatch for {token_name}: " f"expected {expected_id}, got {actual_id}" ) # ============================================================ # LOAD MODEL CONFIG # ============================================================ # model_config.json was written directly from: # # asdict(model_config) # # in the training notebook. with CONFIG_PATH.open("r", encoding="utf-8") as file: CONFIG_DATA = json.load(file) # Build the exact config object expected by model.py. config = LiquidModelConfig(**CONFIG_DATA) # Verify the config agrees with the actual tokenizer. # Your trained checkpoint has vocab_size = 8000. actual_vocab_size = tokenizer.get_vocab_size() if config.vocab_size != actual_vocab_size: raise RuntimeError( f"Vocabulary mismatch: model expects {config.vocab_size}, " f"but tokenizer contains {actual_vocab_size} tokens." ) # Verify the model config is using the same control-token IDs # the tokenizer was trained with. if config.pad_token_id != PAD_ID: raise RuntimeError( f"PAD token mismatch: config={config.pad_token_id}, tokenizer={PAD_ID}" ) if config.bos_token_id != BOS_ID: raise RuntimeError( f"BOS token mismatch: config={config.bos_token_id}, tokenizer={BOS_ID}" ) if config.eos_token_id != EOS_ID: raise RuntimeError( f"EOS token mismatch: config={config.eos_token_id}, tokenizer={EOS_ID}" ) # ============================================================ # DEVICE # ============================================================ # On ZeroGPU, `import spaces` patches torch before this line runs. # torch.cuda.is_available() therefore selects the ZeroGPU CUDA # device correctly. # # On an ordinary CPU Space, it falls back to CPU. DEVICE = torch.device( "cuda" if torch.cuda.is_available() else "cpu" ) # ============================================================ # LOAD CHECKPOINT # ============================================================ # The notebook saved the final model like this: # # { # "model_state_dict": model.state_dict(), # "config": asdict(model_config), # "architecture": "TinyLiquidCausalLanguageModel", # } # # Load it onto CPU first so the checkpoint itself is read safely. checkpoint = torch.load( CHECKPOINT_PATH, map_location="cpu", weights_only=False, ) # Make sure this is actually the expected WVY checkpoint. if "model_state_dict" not in checkpoint: raise RuntimeError( "Checkpoint does not contain 'model_state_dict'." ) # The checkpoint also stores its own config. # Compare it against model_config.json instead of blindly trusting # two files that could theoretically come from different runs. checkpoint_config = checkpoint.get("config") if checkpoint_config is not None: if checkpoint_config != CONFIG_DATA: raise RuntimeError( "model_config.json does not match the config stored " "inside tiny_liquid_causal_lm.pt." ) # The notebook stores the architecture name too. checkpoint_architecture = checkpoint.get("architecture") if ( checkpoint_architecture is not None and checkpoint_architecture != "TinyLiquidCausalLanguageModel" ): raise RuntimeError( f"Unexpected checkpoint architecture: " f"{checkpoint_architecture}" ) # Build the exact architecture defined in model.py. model = TinyLiquidCausalLanguageModel(config) # strict=True means EVERY checkpoint parameter must match the # architecture. Missing or unexpected weights cause startup to fail # instead of silently loading the wrong thing. model.load_state_dict( checkpoint["model_state_dict"], strict=True, ) # Inference mode behavior: dropout is disabled. model.eval() # Move the model at module startup. # # This is important for ZeroGPU. Since `spaces` was imported before # torch, Hugging Face can register the CUDA tensors correctly. model.to(DEVICE) # CPU Spaces benefit from limiting the thread count. # ZeroGPU and normal GPU Spaces do not need this branch. if DEVICE.type == "cpu": torch.set_num_threads( max( 1, min( 4, torch.get_num_threads(), ), ) ) # ============================================================ # VERIFY MODEL SIZE # ============================================================ # The actual checkpoint/architecture bundle sent with the Space # contains 3,314,880 parameters. parameter_count = sum( parameter.numel() for parameter in model.parameters() ) EXPECTED_PARAMETER_COUNT = 3_314_880 if parameter_count != EXPECTED_PARAMETER_COUNT: raise RuntimeError( f"Model parameter mismatch: expected " f"{EXPECTED_PARAMETER_COUNT:,}, got {parameter_count:,}" ) # ============================================================ # ORIGINAL PROMPT FORMAT # ============================================================ def format_user_prompt(prompt): """ This is the exact format_user_prompt() behavior from the training notebook. Example: <|user|> who are you? <|assistant|> """ return ( f"{USER_TOKEN}\n" f"{str(prompt).strip()}\n" f"{ASSISTANT_TOKEN}\n" ) # ============================================================ # GENERATION # ============================================================ # ZeroGPU only supplies a real GPU while a @spaces.GPU-decorated # function is executing. # # This is the function Gradio calls directly, so Hugging Face can # detect the GPU function during startup. @spaces.GPU(duration=120) @torch.inference_mode() def generate_answer( message, history, temperature=0.1, top_k=40, max_new_tokens=2560, ): """ Generate one WVY response. The model-generation logic below follows the actual notebook: 1. format_user_prompt(message) 2. prepend BOS 3. keep the latest 256 tokens 4. run the model 5. block PAD 6. temperature sampling / greedy decoding 7. top-k = 40 by default 8. stop at EOS 9. decode only newly generated tokens `history` exists because Gradio ChatInterface passes it. The original notebook's interactive chat loop did NOT feed previous turns back into generate_answer(), so the model input intentionally remains stateless here too. """ # Normalize the UI input. question = str(message).strip() # An empty message should not be sent through the model. if not question: return "" # -------------------------------------------------------- # EXACT USER/ASSISTANT PROMPT # -------------------------------------------------------- # This produces: # # <|user|> # message # <|assistant|> prompt = format_user_prompt(question) # The original notebook explicitly prepended BOS_ID instead # of putting "<|bos|>" into the string. generated_ids = [ BOS_ID ] + tokenizer.encode(prompt).ids # Remember where generation starts so only the assistant's # newly generated tokens are decoded at the end. prompt_length = len(generated_ids) # Make the Gradio values ordinary Python numeric types. temperature = float(temperature) top_k = int(top_k) max_new_tokens = int(max_new_tokens) # -------------------------------------------------------- # AUTOREGRESSIVE TOKEN LOOP # -------------------------------------------------------- for _ in range(max_new_tokens): # The model was trained with SEQUENCE_LENGTH = 256. # # If the prompt + generated answer grows beyond that, # only the newest 256 tokens are supplied to the model. model_input = torch.tensor( [ generated_ids[ -SEQUENCE_LENGTH: ] ], dtype=torch.long, device=DEVICE, ) # Run the exact custom WVY model. # # Shape before indexing: # # [batch, sequence, vocabulary] model_output = model( model_input ) # We only need logits for the NEXT token after the # final token currently in the sequence. logits = model_output[ "logits" ][0, -1].float() # The notebook explicitly prevented <|pad|> from being # sampled during generation. logits[PAD_ID] = -float("inf") # ---------------------------------------------------- # TOKEN SELECTION # ---------------------------------------------------- # temperature <= 0 reproduces the notebook's greedy path. if temperature <= 0: next_token = int( torch.argmax( logits ).item() ) else: # Temperature rescales logits before sampling. logits = logits / temperature # The notebook uses top-k sampling when top_k > 0. if top_k > 0: # Never request more tokens than the vocabulary has. k = min( top_k, logits.numel(), ) # Keep only the k highest-scoring next tokens. top_values, top_indices = torch.topk( logits, k=k, ) # Convert those logits to probabilities. probabilities = torch.softmax( top_values, dim=-1, ) # Sample one position from the top-k probabilities. sampled_position = torch.multinomial( probabilities, num_samples=1, ) # Convert the sampled position back to the actual # tokenizer token ID. next_token = int( top_indices[ sampled_position ].item() ) else: # This preserves the notebook's fallback path: # sample from the entire vocabulary when top_k # is disabled. probabilities = torch.softmax( logits, dim=-1, ) next_token = int( torch.multinomial( probabilities, num_samples=1, ).item() ) # ---------------------------------------------------- # EOS # ---------------------------------------------------- # The original notebook stops generation when <|eos|> # is predicted. # # EOS itself is NOT appended to generated_ids. if next_token == EOS_ID: break # Add the predicted token to the running sequence. generated_ids.append( next_token ) # ======================================================== # DECODE ASSISTANT OUTPUT ONLY # ======================================================== # Remove the original BOS + user prompt portion. response_ids = generated_ids[ prompt_length: ] # skip_special_tokens=True strips control tokens such as: # # <|pad|> # <|unk|> when marked special by the tokenizer # <|bos|> # <|eos|> # <|user|> # <|assistant|> # <|system|> # # from the displayed response. response = tokenizer.decode( response_ids, skip_special_tokens=True, ).strip() return response # ============================================================ # GRADIO UI # ============================================================ # CSS is passed to launch() because the Gradio version used by # the Space moved `css` out of the Blocks constructor. CSS = """ .gradio-container { max-width: 920px !important; margin: 0 auto !important; } footer { display: none !important; } """ # Build the outer Space layout. with gr.Blocks( title="WVY Tiny Liquid LM" ) as demo: # Space title / architecture information. gr.Markdown( "# WVY Tiny Liquid LM\n" "3.31M-parameter custom causal language model · " "4 liquid-style state-mixer blocks · " "256-token context" ) # No `type="messages"` is supplied here. # # The runtime that produced your error does not accept that # Chatbot constructor argument. chatbot = gr.Chatbot( height=560, ) # These generation controls are created with render=False # because ChatInterface will render them inside its # "Generation settings" accordion. temperature = gr.Slider( minimum=0.0, maximum=1.5, value=0.1, step=0.05, label="Temperature", render=False, ) # This keeps the notebook's default top_k = 40. top_k = gr.Slider( minimum=0, maximum=100, value=40, step=1, label="Top-k", render=False, ) # The current notebook's generate_answer() uses # max_new_tokens = 560. max_new_tokens = gr.Slider( minimum=8, maximum=560, value=560, step=8, label="Max new tokens", render=False, ) # ChatInterface handles the display history itself. # # The callback only has to RETURN ONE STRING. # # That prevents the dict/tuple history mismatch that can happen # when manually updating gr.Chatbot from Blocks callbacks. gr.ChatInterface( fn=generate_answer, chatbot=chatbot, textbox=gr.Textbox( placeholder="Message WVY...", show_label=False, lines=1, ), additional_inputs=[ temperature, top_k, max_new_tokens, ], additional_inputs_accordion="Generation settings", ) # ============================================================ # START SPACE # ============================================================ if __name__ == "__main__": # Queue one model generation at a time. # # The actual GPU allocation is handled by @spaces.GPU above. demo.queue( default_concurrency_limit=1 ).launch( css=CSS )