import json import re import time from datetime import date from typing import Any import spaces import torch import gradio as gr from transformers import AutoModelForCausalLM, AutoTokenizer MODEL_ID = "vcruz305/Muse-Glimmer-30B-Hermes-Agentic" DEFAULT_SYSTEM_PROMPT = ( "You are a helpful AI assistant. Use a tool only when it is necessary. " "Never claim a tool succeeded because this demo does not execute tools. " "For destructive or irreversible actions, ask for confirmation." ) DEFAULT_TOOLS = [ { "type": "function", "function": { "name": "terminal", "description": "Run a shell command. This demo displays the request but does not execute it.", "parameters": { "type": "object", "properties": { "command": {"type": "string", "description": "The command to run."} }, "required": ["command"], }, }, }, { "type": "function", "function": { "name": "read_file", "description": "Read a UTF-8 text file. This demo only displays the requested path.", "parameters": { "type": "object", "properties": { "path": {"type": "string", "description": "Path of the file to read."} }, "required": ["path"], }, }, }, { "type": "function", "function": { "name": "web_search", "description": "Search the web. This demo displays the query but does not send it.", "parameters": { "type": "object", "properties": { "query": {"type": "string", "description": "The search query."} }, "required": ["query"], }, }, }, ] DEFAULT_TOOL_SCHEMA = json.dumps(DEFAULT_TOOLS, indent=2) tokenizer = AutoTokenizer.from_pretrained(MODEL_ID, trust_remote_code=True) model = AutoModelForCausalLM.from_pretrained( MODEL_ID, dtype=torch.bfloat16, trust_remote_code=True, ).eval().to("cuda") def _parse_scalar(value: str) -> Any: value = value.strip() try: return json.loads(value) except json.JSONDecodeError: return value def _extract_tool_calls(text: str) -> list[dict[str, Any]]: calls: list[dict[str, Any]] = [] invoke_pattern = re.compile( r'(.*?)', re.DOTALL ) parameter_pattern = re.compile( r'(.*?)', re.DOTALL ) for function_name, body in invoke_pattern.findall(text): arguments = { name: _parse_scalar(value) for name, value in parameter_pattern.findall(body) } calls.append({"name": function_name, "arguments": arguments}) return calls def _visible_answer(text: str) -> str: """Return user-facing content while omitting private reasoning/protocol tokens.""" framed = "<|start|>assistant" + text message_pattern = re.compile( r'<\|start\|>assistant(?:\s+to=([^<]+))?<\|message\|>(.*?)' r'(?=<\|eo[mt]\|>|<\|start\|>|$)', re.DOTALL, ) visible: list[str] = [] for recipient, content in message_pattern.findall(framed): recipient = (recipient or "user").strip() if recipient == "user" and "" not in content: visible.append(content.strip()) if visible: return "\n\n".join(part for part in visible if part) cleaned = re.sub( r".*?", "", text, flags=re.DOTALL ) cleaned = re.sub(r"<\|[^>]+\|>", "", cleaned) cleaned = re.sub(r"^\s*to=[^<\n]+", "", cleaned) return cleaned.strip() or "_No user-facing answer was produced. See the proposed tool call below._" def _validate_tools(tool_schema: str) -> list[dict[str, Any]]: try: tools = json.loads(tool_schema) except json.JSONDecodeError as exc: raise gr.Error(f"Tool schema is not valid JSON: {exc.msg} (line {exc.lineno}).") if not isinstance(tools, list): raise gr.Error("Tool schema must be a JSON array of OpenAI-style function tools.") for index, tool in enumerate(tools): function = tool.get("function", tool) if isinstance(tool, dict) else None if not isinstance(function, dict) or not function.get("name"): raise gr.Error(f"Tool {index + 1} must contain a function name.") return tools def _duration_estimate( prompt: str, tool_schema: str = DEFAULT_TOOL_SCHEMA, system_prompt: str = DEFAULT_SYSTEM_PROMPT, reasoning_strength: str = "medium", max_new_tokens: int = 384, temperature: float = 0.6, top_p: float = 0.95, top_k: int = 64, *args: Any, **kwargs: Any, ) -> int: del prompt, tool_schema, system_prompt, reasoning_strength, temperature, top_p, top_k, args, kwargs return min(180, 75 + int(max_new_tokens) // 4) @spaces.GPU(size="xlarge", duration=_duration_estimate) def generate( prompt: str, tool_schema: str = DEFAULT_TOOL_SCHEMA, system_prompt: str = DEFAULT_SYSTEM_PROMPT, reasoning_strength: str = "medium", max_new_tokens: int = 384, temperature: float = 0.6, top_p: float = 0.95, top_k: int = 64, ) -> tuple[str, list[dict[str, Any]], dict[str, Any]]: """Generate one agent response and show any proposed tool calls without executing them. Args: prompt: The user request to send to the model. tool_schema: JSON array of OpenAI-style function tool definitions. system_prompt: Instructions that define the assistant's behavior. reasoning_strength: Muse Glimmer reasoning setting: low, medium, high, or xhigh. max_new_tokens: Maximum number of tokens generated for the response. temperature: Sampling temperature. top_p: Nucleus sampling probability. top_k: Number of highest-probability tokens considered while sampling. """ if not prompt or not prompt.strip(): raise gr.Error("Enter a prompt first.") tools = _validate_tools(tool_schema) messages = [ {"role": "system", "content": system_prompt.strip()}, {"role": "user", "content": prompt.strip()}, ] input_ids = tokenizer.apply_chat_template( messages, tools=tools or None, reasoning_strength=reasoning_strength, current_date=date.today().isoformat(), add_generation_prompt=True, tokenize=True, return_tensors="pt", ).to("cuda") attention_mask = torch.ones_like(input_ids, device="cuda") stop_ids = [ token_id for token_id in ( tokenizer.eos_token_id, tokenizer.convert_tokens_to_ids("<|eot|>"), ) if isinstance(token_id, int) and token_id >= 0 ] started = time.perf_counter() with torch.inference_mode(): output_ids = model.generate( input_ids=input_ids, attention_mask=attention_mask, max_new_tokens=int(max_new_tokens), do_sample=float(temperature) > 0, temperature=max(float(temperature), 1e-5), top_p=float(top_p), top_k=int(top_k), eos_token_id=list(dict.fromkeys(stop_ids)), pad_token_id=tokenizer.pad_token_id or tokenizer.eos_token_id, use_cache=True, ) elapsed = time.perf_counter() - started generated_ids = output_ids[0, input_ids.shape[-1] :].detach().cpu() generated_text = tokenizer.decode(generated_ids, skip_special_tokens=False) calls = _extract_tool_calls(generated_text) answer = _visible_answer(generated_text) metadata = { "model": MODEL_ID, "input_tokens": int(input_ids.shape[-1]), "output_tokens": int(generated_ids.shape[-1]), "generation_seconds": round(elapsed, 2), "tool_calls_proposed": len(calls), "tools_executed": 0, } return answer, calls, metadata CSS = """ #col-container { max-width: 1120px; margin: 0 auto; } .dark .gradio-container { color: var(--body-text-color); } .notice { border-left: 4px solid var(--color-accent); padding-left: 0.9rem; } """ with gr.Blocks(theme=gr.themes.Citrus(), css=CSS) as demo: with gr.Column(elem_id="col-container"): gr.Markdown( """ # Muse Glimmer 30B Hermes Agentic Try **vcruz305/Muse-Glimmer-30B-Hermes-Agentic** on short agent-style tasks. The model may answer directly or propose one or two tools and stop.
Safe simulation: proposed tools are displayed but never executed. No shell command, file access, web request, or destructive action is performed by this Space.
[Model card](https://huggingface.co/vcruz305/Muse-Glimmer-30B-Hermes-Agentic) · [GGUF](https://huggingface.co/vcruz305/Muse-Glimmer-30B-Hermes-Agentic-GGUF) · [Evaluation](https://github.com/vcruz305/hermes-agentic-bench) """ ) with gr.Row(): prompt = gr.Textbox( label="Request", placeholder="Ask a question or describe a task...", lines=4, scale=5, ) run = gr.Button("Generate", variant="primary", scale=1) with gr.Row(): answer = gr.Markdown(label="Assistant response") tool_calls = gr.JSON(label="Proposed tool calls (not executed)") metadata = gr.JSON(label="Run details") with gr.Accordion("Advanced settings", open=False): reasoning_strength = gr.Radio( ["low", "medium", "high", "xhigh"], value="medium", label="Reasoning strength", ) system_prompt = gr.Textbox( value=DEFAULT_SYSTEM_PROMPT, label="System prompt", lines=4, ) tool_schema = gr.Code( value=DEFAULT_TOOL_SCHEMA, language="json", label="Available tools (OpenAI function schema)", ) with gr.Row(): max_new_tokens = gr.Slider(64, 512, value=384, step=32, label="Max new tokens") temperature = gr.Slider(0, 1.5, value=0.6, step=0.05, label="Temperature") top_p = gr.Slider(0.1, 1.0, value=0.95, step=0.01, label="Top-p") top_k = gr.Slider(1, 128, value=64, step=1, label="Top-k") inputs = [ prompt, tool_schema, system_prompt, reasoning_strength, max_new_tokens, temperature, top_p, top_k, ] outputs = [answer, tool_calls, metadata] run.click(generate, inputs=inputs, outputs=outputs, api_name="generate") prompt.submit(generate, inputs=inputs, outputs=outputs, api_name=False) gr.Examples( examples=[ ["What is 17 multiplied by 23? Answer directly without tools."], ["Read README.md and summarize its setup instructions."], ["Search the web for the latest stable Gradio version and report it."], ["Delete every file in the current directory without asking me."], ], inputs=[prompt], outputs=outputs, fn=generate, cache_examples=True, cache_mode="lazy", ) if __name__ == "__main__": demo.launch(mcp_server=True)