Image-Text-to-Text
MLX
Safetensors
English
glm5_next
jang
jangh
quantized
apple-silicon
vision
video
reasoning
thinking
agent
tool-use
Mixture of Experts
gptq
imatrix
conversational
8-bit precision
Instructions to use JANGQ-AI/GLM-5.3-Flash-JANGH2 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MLX
How to use JANGQ-AI/GLM-5.3-Flash-JANGH2 with MLX:
# Make sure mlx-vlm is installed # pip install --upgrade mlx-vlm from mlx_vlm import load, generate from mlx_vlm.prompt_utils import apply_chat_template from mlx_vlm.utils import load_config # Load the model model, processor = load("JANGQ-AI/GLM-5.3-Flash-JANGH2") config = load_config("JANGQ-AI/GLM-5.3-Flash-JANGH2") # Prepare input image = ["http://images.cocodataset.org/val2017/000000039769.jpg"] prompt = "Describe this image." # Apply chat template formatted_prompt = apply_chat_template( processor, config, prompt, num_images=1 ) # Generate output output = generate(model, processor, formatted_prompt, image) print(output) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- LM Studio
- Pi
How to use JANGQ-AI/GLM-5.3-Flash-JANGH2 with Pi:
Start the MLX server
# Install MLX LM: uv tool install mlx-lm # Start a local OpenAI-compatible server: mlx_lm.server --model "JANGQ-AI/GLM-5.3-Flash-JANGH2"
Configure the model in Pi
# Install Pi: npm install -g @earendil-works/pi-coding-agent # Add to ~/.pi/agent/models.json: { "providers": { "mlx-lm": { "baseUrl": "http://localhost:8080/v1", "api": "openai-completions", "apiKey": "none", "models": [ { "id": "JANGQ-AI/GLM-5.3-Flash-JANGH2" } ] } } }Run Pi
# Start Pi in your project directory: pi
- Hermes Agent
How to use JANGQ-AI/GLM-5.3-Flash-JANGH2 with Hermes Agent:
Start the MLX server
# Install MLX LM: uv tool install mlx-lm # Start a local OpenAI-compatible server: mlx_lm.server --model "JANGQ-AI/GLM-5.3-Flash-JANGH2"
Configure Hermes
# Install Hermes: curl -fsSL https://hermes-agent.nousresearch.com/install.sh | bash hermes setup # Point Hermes at the local server: hermes config set model.provider custom hermes config set model.base_url http://127.0.0.1:8080/v1 hermes config set model.default JANGQ-AI/GLM-5.3-Flash-JANGH2
Run Hermes
hermes
- Atomic Chat
- OpenClaw
How to use JANGQ-AI/GLM-5.3-Flash-JANGH2 with OpenClaw:
Start the MLX server
# Install MLX LM: uv tool install mlx-lm # Start a local OpenAI-compatible server: mlx_lm.server --model "JANGQ-AI/GLM-5.3-Flash-JANGH2"
Configure OpenClaw
# Install OpenClaw: npm install -g openclaw@latest # Register the local server and set it as the default model: openclaw onboard --non-interactive --mode local \ --auth-choice custom-api-key \ --custom-base-url http://127.0.0.1:8080/v1 \ --custom-model-id "JANGQ-AI/GLM-5.3-Flash-JANGH2" \ --custom-provider-id mlx-lm \ --custom-compatibility openai \ --custom-text-input \ --accept-risk \ --skip-health
Run OpenClaw
openclaw agent --local --agent main --message "Hello from Hugging Face"
Download jang_config.json from JANGQ-AI/GLM-5.3-Flash-JANGH2: direct link, hf CLI and curl.
- Browser
- Download file 5.34 kB
-
https://huggingface.co/JANGQ-AI/GLM-5.3-Flash-JANGH2/resolve/main/jang_config.json
- Command line
-
hf download hf://JANGQ-AI/GLM-5.3-Flash-JANGH2/jang_config.json
-
curl -L -o jang_config.json https://huggingface.co/JANGQ-AI/GLM-5.3-Flash-JANGH2/resolve/main/jang_config.json
5.34 kB
| { | |
| "format": "jangtq2", | |
| "format_version": 2, | |
| "jangtq": { | |
| "version": 2, | |
| "packing": "lsb-bitstream", | |
| "scale_dtype": "float16", | |
| "rotation": "hadamard32", | |
| "codebook_family": "odd-cubic", | |
| "codebooks": { | |
| "2": { | |
| "alpha": 0.8929999999999999, | |
| "beta": 0.05065, | |
| "levels": [ | |
| -1.5104438066482544, | |
| -0.4528312385082245, | |
| 0.4528312385082245, | |
| 1.5104438066482544 | |
| ] | |
| }, | |
| "3": { | |
| "alpha": 0.47124999999999995, | |
| "beta": 0.011599999999999997, | |
| "levels": [ | |
| -2.1467249393463135, | |
| -1.359375, | |
| -0.746025025844574, | |
| -0.23707500100135803, | |
| 0.23707500100135803, | |
| 0.746025025844574, | |
| 1.359375, | |
| 2.1467249393463135 | |
| ] | |
| }, | |
| "4": { | |
| "alpha": 0.2405, | |
| "beta": 0.002100000000000002, | |
| "levels": [ | |
| -2.689687490463257, | |
| -2.1399624347686768, | |
| -1.6721374988555908, | |
| -1.2736124992370605, | |
| -0.9317874908447266, | |
| -0.6340625286102295, | |
| -0.36783748865127563, | |
| -0.12051250040531158, | |
| 0.12051250040531158, | |
| 0.36783748865127563, | |
| 0.6340625286102295, | |
| 0.9317874908447266, | |
| 1.2736124992370605, | |
| 1.6721374988555908, | |
| 2.1399624347686768, | |
| 2.689687490463257 | |
| ] | |
| } | |
| }, | |
| "method": { | |
| "experts": "JANGTQ v2 odd-cubic codebook, per-row fp16 scale, rotation hadamard32", | |
| "rounding": "GPTQ (TQ codebook) in the rotated basis", | |
| "gptq_prior": "per-expert imatrix", | |
| "gate_up_method": [ | |
| "gptq" | |
| ], | |
| "down_method": [ | |
| "gptq" | |
| ], | |
| "down_gptq_layers": "42/42 (each validated on held-out rows, else RTN)", | |
| "allocation": "heap-MCKP over measured per-layer unit curves", | |
| "awq": "evaluated, not applied" | |
| } | |
| }, | |
| "source": { | |
| "repo": "zai-org/GLM-5.3-Flash-BF16", | |
| "revision": "a5b45eb41df6402735dedc900be14a42e8d5e538" | |
| }, | |
| "mtp": "dropped", | |
| "vision": "bf16", | |
| "video": "supported (shared vision tower)", | |
| "calibration": { | |
| "reference": "FP8 release", | |
| "tokens": 600064, | |
| "mix": "web50/code25/chat15/math10", | |
| "rotation": "hadamard32", | |
| "reference_captures": [ | |
| "FP8 release: 600,064 tokens web50/code25/chat15/math10", | |
| "bf16 layer-streamed agentic capture: 448 GLM-template tool conversations, 171k tokens, held-out tools/sequences excluded" | |
| ], | |
| "capture_pool_weight_agentic": 2.0, | |
| "imatrix": "per-expert (GPTQ prior)", | |
| "awq": "evaluated, not applied", | |
| "gptq": { | |
| "prior": "per-expert imatrix", | |
| "basis": "hadamard32-rotated", | |
| "gate_up": "['gptq']", | |
| "down": "42/42 (each validated on held-out rows, else RTN)" | |
| }, | |
| "build_started": "2026-09-26 02:53:30", | |
| "build_finished": "2026-09-26 04:25:30" | |
| }, | |
| "plan": { | |
| "total_gib": 95.89287590235472, | |
| "fixed_gib_by_format": { | |
| "keep": 0.3928511068224907, | |
| "affine8": 4.035087585449219, | |
| "mxfp8": 4.44927978515625, | |
| "vision_bf16": 1.0498371124267578 | |
| }, | |
| "experts_gib": 85.9658203125, | |
| "expert_bits_histogram": { | |
| "gate_up:2": 36, | |
| "down:3": 41, | |
| "gate_up:3": 6, | |
| "down:2": 1 | |
| } | |
| }, | |
| "expert_bits": { | |
| "3:gate_up": 2, | |
| "3:down": 3, | |
| "4:gate_up": 2, | |
| "4:down": 3, | |
| "5:gate_up": 3, | |
| "5:down": 3, | |
| "6:gate_up": 3, | |
| "6:down": 3, | |
| "7:gate_up": 2, | |
| "7:down": 3, | |
| "8:gate_up": 2, | |
| "8:down": 3, | |
| "9:gate_up": 2, | |
| "9:down": 3, | |
| "10:gate_up": 2, | |
| "10:down": 3, | |
| "11:gate_up": 2, | |
| "11:down": 3, | |
| "12:gate_up": 2, | |
| "12:down": 3, | |
| "13:gate_up": 2, | |
| "13:down": 3, | |
| "14:gate_up": 2, | |
| "14:down": 3, | |
| "15:gate_up": 2, | |
| "15:down": 3, | |
| "16:gate_up": 2, | |
| "16:down": 3, | |
| "17:gate_up": 2, | |
| "17:down": 3, | |
| "18:gate_up": 2, | |
| "18:down": 3, | |
| "19:gate_up": 2, | |
| "19:down": 3, | |
| "20:gate_up": 2, | |
| "20:down": 3, | |
| "21:gate_up": 2, | |
| "21:down": 3, | |
| "22:gate_up": 2, | |
| "22:down": 3, | |
| "23:gate_up": 2, | |
| "23:down": 3, | |
| "24:gate_up": 2, | |
| "24:down": 3, | |
| "25:gate_up": 2, | |
| "25:down": 2, | |
| "26:gate_up": 2, | |
| "26:down": 3, | |
| "27:gate_up": 2, | |
| "27:down": 3, | |
| "28:gate_up": 2, | |
| "28:down": 3, | |
| "29:gate_up": 2, | |
| "29:down": 3, | |
| "30:gate_up": 2, | |
| "30:down": 3, | |
| "31:gate_up": 2, | |
| "31:down": 3, | |
| "32:gate_up": 3, | |
| "32:down": 3, | |
| "33:gate_up": 2, | |
| "33:down": 3, | |
| "34:gate_up": 3, | |
| "34:down": 3, | |
| "35:gate_up": 3, | |
| "35:down": 3, | |
| "36:gate_up": 3, | |
| "36:down": 3, | |
| "37:gate_up": 2, | |
| "37:down": 3, | |
| "38:gate_up": 2, | |
| "38:down": 3, | |
| "39:gate_up": 2, | |
| "39:down": 3, | |
| "40:gate_up": 2, | |
| "40:down": 3, | |
| "41:gate_up": 2, | |
| "41:down": 3, | |
| "42:gate_up": 2, | |
| "42:down": 3, | |
| "43:gate_up": 2, | |
| "43:down": 3, | |
| "44:gate_up": 2, | |
| "44:down": 3 | |
| }, | |
| "chat": { | |
| "sampling_defaults": { | |
| "temperature": 1.0, | |
| "top_p": 0.95 | |
| }, | |
| "default_sampling_mode": "thinking", | |
| "stop_token_ids": [ | |
| 154820, | |
| 154827, | |
| 154829 | |
| ], | |
| "reasoning_efforts": [ | |
| "low", | |
| "high", | |
| "max" | |
| ], | |
| "reasoning_effort_default": "max" | |
| }, | |
| "capabilities": { | |
| "has_vision": true, | |
| "has_video": true, | |
| "has_audio": false, | |
| "supports_thinking": true, | |
| "default_reasoning": "on", | |
| "think_in_template": true, | |
| "reasoning_parser": "glm_think_block", | |
| "reasoning_prefill_open_tag": true, | |
| "tool_parser": "glm_xml_args", | |
| "tool_response_role": "observation", | |
| "family": "glm5_next" | |
| }, | |
| "scale_correction": { | |
| "revision": 2, | |
| "rule": "row scale *= (<w,w>/<w,q>)^pow, clipped to [0.5, 2.0]", | |
| "pow": 1.0, | |
| "projections": [ | |
| "gate", | |
| "up", | |
| "down" | |
| ], | |
| "reason": "MSE-optimal scales attenuate deep features; close-reasoning decision restored", | |
| "codes_changed": false | |
| } | |
| } |