Instructions to use ukisai/Swift-1.5-4bit-MLX with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MLX
How to use ukisai/Swift-1.5-4bit-MLX with MLX:
# Make sure mlx-lm is installed # pip install --upgrade mlx-lm # Generate text with mlx-lm from mlx_lm import load, generate model, tokenizer = load("ukisai/Swift-1.5-4bit-MLX") prompt = "Write a story about Einstein" messages = [{"role": "user", "content": prompt}] prompt = tokenizer.apply_chat_template( messages, add_generation_prompt=True ) text = generate(model, tokenizer, prompt=prompt, verbose=True) - Notebooks
- Google Colab
- Kaggle
- Local Apps Settings
- LM Studio
- Pi
How to use ukisai/Swift-1.5-4bit-MLX with Pi:
Start the MLX server
# Install MLX LM: uv tool install mlx-lm # Start a local OpenAI-compatible server: mlx_lm.server --model "ukisai/Swift-1.5-4bit-MLX"
Configure the model in Pi
# Install Pi: npm install -g @earendil-works/pi-coding-agent # Add to ~/.pi/agent/models.json: { "providers": { "mlx-lm": { "baseUrl": "http://localhost:8080/v1", "api": "openai-completions", "apiKey": "none", "models": [ { "id": "ukisai/Swift-1.5-4bit-MLX" } ] } } }Run Pi
# Start Pi in your project directory: pi
- MLX LM
How to use ukisai/Swift-1.5-4bit-MLX with MLX LM:
Generate or start a chat session
# Install MLX LM uv tool install mlx-lm # Interactive chat REPL mlx_lm.chat --model "ukisai/Swift-1.5-4bit-MLX"
Run an OpenAI-compatible server
# Install MLX LM uv tool install mlx-lm # Start the server mlx_lm.server --model "ukisai/Swift-1.5-4bit-MLX" # Calling the OpenAI-compatible server with curl curl -X POST "http://localhost:8000/v1/chat/completions" \ -H "Content-Type: application/json" \ --data '{ "model": "ukisai/Swift-1.5-4bit-MLX", "messages": [ {"role": "user", "content": "Hello"} ] }' - Hermes Agent
How to use ukisai/Swift-1.5-4bit-MLX with Hermes Agent:
Start the MLX server
# Install MLX LM: uv tool install mlx-lm # Start a local OpenAI-compatible server: mlx_lm.server --model "ukisai/Swift-1.5-4bit-MLX"
Configure Hermes
# Install Hermes: curl -fsSL https://hermes-agent.nousresearch.com/install.sh | bash hermes setup # Point Hermes at the local server: hermes config set model.provider custom hermes config set model.base_url http://127.0.0.1:8080/v1 hermes config set model.default ukisai/Swift-1.5-4bit-MLX
Run Hermes
hermes
- Atomic Chat
- OpenClaw
How to use ukisai/Swift-1.5-4bit-MLX with OpenClaw:
Start the MLX server
# Install MLX LM: uv tool install mlx-lm # Start a local OpenAI-compatible server: mlx_lm.server --model "ukisai/Swift-1.5-4bit-MLX"
Configure OpenClaw
# Install OpenClaw: npm install -g openclaw@latest # Register the local server and set it as the default model: openclaw onboard --non-interactive --mode local \ --auth-choice custom-api-key \ --custom-base-url http://127.0.0.1:8080/v1 \ --custom-model-id "ukisai/Swift-1.5-4bit-MLX" \ --custom-provider-id mlx-lm \ --custom-compatibility openai \ --custom-text-input \ --accept-risk \ --skip-health
Run OpenClaw
openclaw agent --local --agent main --message "Hello from Hugging Face"
Download compatibility/validate-quant.py from ukisai/Swift-1.5-4bit-MLX: direct link, hf CLI and curl.
- Browser
- Download file 8.62 kB
-
https://huggingface.co/ukisai/Swift-1.5-4bit-MLX/resolve/9fd3d5f738d71f50e0fe84574c44948c0457a01b/compatibility/validate-quant.py
- Command line
-
hf download hf://ukisai/Swift-1.5-4bit-MLX@9fd3d5f738d71f50e0fe84574c44948c0457a01b/compatibility/validate-quant.py
-
curl -L -o validate-quant.py https://huggingface.co/ukisai/Swift-1.5-4bit-MLX/resolve/9fd3d5f738d71f50e0fe84574c44948c0457a01b/compatibility/validate-quant.py
8.62 kB
| """Validate the saved real Swift MLX artifact, including all components.""" | |
| import hashlib | |
| import json | |
| import platform | |
| import resource | |
| import time | |
| from collections import Counter | |
| from datetime import datetime, timezone | |
| from pathlib import Path | |
| import os | |
| import mlx.core as mx | |
| from mlx.utils import tree_flatten | |
| from mlx_lm import load, stream_generate | |
| from mlx_lm.sample_utils import make_sampler | |
| from transformers import AutoProcessor, AutoTokenizer | |
| from PIL import Image | |
| root = Path(__file__).resolve().parents[1] | |
| source = Path(os.environ['SWIFT_SOURCE_DIR']).resolve(strict=True) | |
| output = Path(os.environ.get('SWIFT_MLX_OUTPUT', root / 'Swift-1.5-4bit-MLX')).resolve(strict=True) | |
| logs = Path(os.environ.get('SWIFT_VALIDATION_DIR', root / 'validation-output')).resolve(strict=True) | |
| verified = json.loads((logs / 'source-verification.json').read_text()) | |
| conversion = json.loads((logs / 'conversion-result.json').read_text()) | |
| assert conversion['returncode'] == 0 | |
| assert not (logs / 'quant-validation-results.json').exists() | |
| mx.set_default_device(mx.cpu) | |
| start = time.monotonic() | |
| model, tokenizer, config = load(str(output), lazy=False, return_config=True) | |
| load_seconds = time.monotonic() - start | |
| load_memory = mx.get_active_memory() | |
| parameters = dict(tree_flatten(model.parameters())) | |
| assert type(model).__module__ == 'mlx_lm.models.qwen3_5_full' | |
| assert config['quantization'] == {'mode': 'affine', 'bits': 4, 'group_size': 64} | |
| assert config['vision_config'] == verified['config']['vision_config'] | |
| assert config['text_config'] == verified['config']['text_config'] | |
| assert config['tie_word_embeddings'] == verified['config']['tie_word_embeddings'] | |
| rows = model.weight_mapping() | |
| assert {r['source'] for r in rows} == set(verified['tensors']) | |
| assert len(rows) == 1199 | |
| accounted = set() | |
| for row in rows: | |
| assert list(row['source_shape']) == verified['tensors'][row['source']]['shape'] | |
| name = row['destination'] | |
| assert name in parameters, name | |
| native = [name] | |
| if name.endswith('.weight') and name[:-7]+'.scales' in parameters: | |
| native += [name[:-7]+'.scales', name[:-7]+'.biases'] | |
| assert parameters[name].dtype == mx.uint32 | |
| row['storage'] = 'affine/4-bit/group-size-64' | |
| else: | |
| assert parameters[name].dtype == mx.bfloat16, name | |
| row['storage'] = 'original BF16, with the documented layout mapping' | |
| row['saved_tensors'] = native | |
| accounted.update(native) | |
| assert accounted == set(parameters) | |
| categories = dict(Counter(r['category'] for r in rows)) | |
| assert categories == {'text': 851, 'MTP': 15, 'vision': 333} | |
| for name, value in parameters.items(): | |
| if mx.issubdtype(value.dtype, mx.floating): | |
| assert bool(mx.all(mx.isfinite(value))), f'Nonfinite values in {name}' | |
| print(f'Loaded {len(parameters)} saved tensors; all 1199 source tensors accounted for.', flush=True) | |
| # Compare every unquantized source tensor bit-for-bit after its required layout change. | |
| unchanged = [r for r in rows if r['storage'].startswith('original BF16')] | |
| for shard in sorted({verified['tensors'][r['source']]['shard'] for r in unchanged}): | |
| raw = mx.load(str(source / shard)) | |
| for row in unchanged: | |
| if verified['tensors'][row['source']]['shard'] != shard: | |
| continue | |
| value = raw[row['source']] | |
| if row['transform'] == 'transpose(0,2,1)': | |
| value = value.transpose(0, 2, 1) | |
| elif row['transform'] == 'transpose(0,2,3,4,1)': | |
| value = value.transpose(0, 2, 3, 4, 1) | |
| assert bool(mx.all(value == parameters[row['destination']])), row['source'] | |
| del raw | |
| print(f'All {len(unchanged)} unquantized tensors equal the original BF16 values.', flush=True) | |
| assets = {} | |
| for name in ['generation_config.json', *model.extra_save_files]: | |
| if (source / name).is_file(): | |
| assert (source / name).read_bytes() == (output / name).read_bytes(), name | |
| assets[name] = hashlib.sha256((output / name).read_bytes()).hexdigest() | |
| hf_tokenizer = AutoTokenizer.from_pretrained(output, local_files_only=True, trust_remote_code=False) | |
| processor = AutoProcessor.from_pretrained(output, local_files_only=True, trust_remote_code=False) | |
| chats = [] | |
| for options in ({'enable_thinking': False}, {'reasoning_effort': 'low'}, {'reasoning_effort': 'xhigh'}): | |
| prompt = hf_tokenizer.apply_chat_template([{'role': 'user', 'content': 'Say hello.'}], tokenize=False, add_generation_prompt=True, **options) | |
| assert prompt and hf_tokenizer.encode(prompt, add_special_tokens=False) | |
| chats.append({'options': options, 'rendered': prompt}) | |
| mapping = {'source_tensors':1199, 'mapped_source_tensors':1199, 'native_tensors':len(parameters), 'ignored':0, 'unexplained':0, 'rows':rows} | |
| mapping_path = logs / 'quant-tensor-mapping-manifest.json' | |
| if mapping_path.exists(): | |
| assert json.loads(mapping_path.read_text()) == json.loads(json.dumps(mapping)) | |
| else: | |
| with mapping_path.open('x') as f: | |
| json.dump(mapping, f, indent=2) | |
| # The official Linux scalar BF16 QMM accumulates in BF16 (8192 ones -> 256). | |
| # Promote only in-memory floating values; packed 4-bit tensors/files stay unchanged. | |
| model.apply(lambda value: value.astype(mx.float32) if mx.issubdtype(value.dtype, mx.floating) else value) | |
| mx.eval(model.parameters()) | |
| print('CPU inference uses FP32 floating values; saved 4-bit weights are unchanged.', flush=True) | |
| prompt = tokenizer.apply_chat_template([{'role':'user','content':'Reply with exactly: Hello from Swift.'}], tokenize=False, add_generation_prompt=True, enable_thinking=False) | |
| pieces, tokens, last = [], [], None | |
| generation_start = time.monotonic() | |
| for response in stream_generate(model, tokenizer, prompt=prompt, max_tokens=24, sampler=make_sampler(temp=0.0), prefill_step_size=64): | |
| assert bool(mx.all(mx.isfinite(response.logprobs))), 'Nonfinite generation probabilities' | |
| pieces.append(response.text); tokens.append(response.token); last = response | |
| print(response.text, end='', flush=True) | |
| generated = ''.join(pieces) | |
| assert generated.strip(), 'Empty text generation' | |
| print('\nText generation passed.', flush=True) | |
| generation = {'prompt':prompt, 'text':generated, 'token_ids':tokens, 'tokens':last.generation_tokens, 'tokens_per_second':last.generation_tps, 'prompt_tokens_per_second':last.prompt_tps, 'elapsed_seconds':time.monotonic()-generation_start, 'finish_reason':last.finish_reason} | |
| ids = mx.array([hf_tokenizer.encode('Hello', add_special_tokens=False)[:2]], dtype=mx.int32) | |
| hidden = model.model(ids) | |
| mtp = model.mtp_logits(ids, hidden) | |
| mx.eval(mtp) | |
| assert bool(mx.all(mx.isfinite(mtp))) | |
| mtp_result = {'status':'PASS', 'shape':list(mtp.shape), 'path':'Explicit MTP step with real text hidden states and shared LM head; speculative generation is not integrated'} | |
| print('Real-weight MTP step passed.', flush=True) | |
| pixels = processor.image_processor(images=[Image.new('RGB', (256,256), (64,128,192))], return_tensors='np') | |
| features = model.visual(mx.array(pixels['pixel_values']), pixels['image_grid_thw']) | |
| mx.eval(features) | |
| assert bool(mx.all(mx.isfinite(features))) | |
| vision_result = {'status':'PASS', 'shape':list(features.shape), 'grid':pixels['image_grid_thw'].tolist(), 'path':'Vision encoder only; image/video insertion and multimodal text generation are not implemented'} | |
| print('Real-weight vision encoder passed.', flush=True) | |
| result = {'status':'PASS', 'recorded_at':datetime.now(timezone.utc).isoformat(), 'source_repo':conversion['source_repo'], 'source_revision':conversion['source_revision'], 'source_shards':18, 'source_shard_bytes':verified['shard_bytes'], 'source_tensors':1199, 'mapped_source_tensors':1199, 'saved_tensors':len(parameters), 'categories':categories, 'ignored_tensors':0, 'unexplained_tensors':0, 'exact_unquantized_tensors':len(unchanged), 'quantization':config['quantization'], 'all_floating_tensors_finite':True, 'tokenizer':type(hf_tokenizer).__name__, 'processor':type(processor).__name__, 'assets_sha256':assets, 'chat_templates':chats, 'load_seconds':load_seconds, 'load_memory_bytes':load_memory, 'process_peak_rss_bytes':resource.getrusage(resource.RUSAGE_SELF).ru_maxrss * (1024 if platform.system()=='Linux' else 1), 'inference_floating_dtype':'float32, CPU runtime only; stored floating tensors remain BF16', 'native_bf16_cpu_inference':'Aborted after reproducing incorrect accumulation in the official Linux BF16 quantized matmul. See cpu-quantized-matmul-diagnostic.json.', 'generation':generation, 'mtp':mtp_result, 'vision':vision_result, 'total_validation_seconds':time.monotonic()-start} | |
| with (logs / 'quant-validation-results.json').open('x') as f:json.dump(result,f,indent=2) | |
| print(json.dumps(result,indent=2),flush=True) | |