""" ██████╗ ██╗ ██╗██████╗ ███████╗ ██╗██████╗ ╚════██╗██║ ██║██╔══██╗██╔════╝ ███║██╔══██╗ █████╔╝██║ ██║██████╔╝█████╗ ╚██║██████╔╝ ╚═══██╗██║ ██║██╔══██╗██╔══╝ ██║██╔══██╗ ██████╔╝╚██████╔╝██║ ██║███████╗ ██║██████╔╝ ╚═════╝ ╚═════╝ ╚═╝ ╚═╝╚══════╝ ╚═╝╚═════╝ Zure 1B Type /exit to quit | /reset to clear history """ import os import sys BANNER = """ \033[95m ██████╗ ██╗ ██╗██████╗ ███████╗ ██╗██████╗ ╚════██╗██║ ██║██╔══██╗██╔════╝ ███║██╔══██╗ █████╔╝██║ ██║██████╔╝█████╗ ╚██║██████╔╝ ╚═══██╗██║ ██║██╔══██╗██╔══╝ ██║██╔══██╗ ██████╔╝╚██████╔╝██║ ██║███████╗ ██║██████╔╝ ╚═════╝ ╚═════╝ ╚═╝ ╚═╝╚══════╝ ╚═╝╚═════╝\033[0m \033[90m Zure 1B Type \033[93m/exit\033[90m to quit | \033[93m/reset\033[90m to clear chat history\033[0m """ GGUF_PATH = os.path.join(os.path.dirname(__file__), "zure_1b.gguf") def load_model(): try: from llama_cpp import Llama except ImportError: print("\033[91mError:\033[0m llama-cpp-python is not installed.") print("Install it with Metal (Apple Silicon GPU) support:\n") print(" \033[93mCMAKE_ARGS=\"-DGGML_METAL=on\" pip install llama-cpp-python\033[0m\n") sys.exit(1) if not os.path.exists(GGUF_PATH): print(f"\033[91mError:\033[0m GGUF model file not found at: {GGUF_PATH}") sys.exit(1) print(f"\033[90mLoading model from {GGUF_PATH}...\033[0m") llm = Llama( model_path=GGUF_PATH, n_ctx=2048, # context window n_threads=4, # CPU threads n_gpu_layers=-1, # offload all layers to Metal GPU (-1 = all) verbose=False, ) return llm def format_prompt(history): """Format chat history into Gemma 3 chat template.""" prompt = "" for role, content in history: if role == "user": prompt += f"user\n{content}\n" else: prompt += f"model\n{content}\n" prompt += "model\n" return prompt def stream_response(llm, history): """Stream the model's response token by token.""" prompt = format_prompt(history) print(f"\n\033[95mZure:\033[0m ", end="", flush=True) full_response = "" for token in llm( prompt, max_tokens=512, temperature=0.75, top_p=0.9, top_k=40, repeat_penalty=1.1, stop=["", ""], stream=True, ): chunk = token["choices"][0]["text"] print(chunk, end="", flush=True) full_response += chunk print() # newline after response return full_response.strip() def main(): print(BANNER) llm = load_model() print(f"\033[92m✓ Model loaded successfully.\033[0m\n") history = [] while True: try: user_input = input("\033[96mYou:\033[0m ").strip() except (EOFError, KeyboardInterrupt): print("\n\033[90mExiting Zure 1B. Goodbye.\033[0m") break if not user_input: continue if user_input.lower() == "/exit": print("\033[90mExiting Zure 1B. Goodbye.\033[0m") break if user_input.lower() == "/reset": history = [] print("\033[90m[Chat history cleared]\033[0m\n") continue history.append(("user", user_input)) response = stream_response(llm, history) history.append(("model", response)) print() if __name__ == "__main__": main()