Instructions to use yocoms/system1-qlora with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- PEFT
How to use yocoms/system1-qlora with PEFT:
Task type is invalid.
- Notebooks
- Google Colab
- Kaggle
System-One QLoRA: 0.6B + 4B adapters, format-matched NLI recipe
Browse files- 06b/adapter_config.json +51 -0
- 06b/adapter_model.safetensors +3 -0
- 4b/adapter_config.json +51 -0
- 4b/adapter_model.safetensors +3 -0
- README.md +83 -0
06b/adapter_config.json
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"alora_invocation_tokens": null,
|
| 3 |
+
"alpha_pattern": {},
|
| 4 |
+
"arrow_config": null,
|
| 5 |
+
"auto_mapping": null,
|
| 6 |
+
"base_model_name_or_path": "Qwen/Qwen3-0.6B",
|
| 7 |
+
"bias": "none",
|
| 8 |
+
"corda_config": null,
|
| 9 |
+
"ensure_weight_tying": false,
|
| 10 |
+
"eva_config": null,
|
| 11 |
+
"exclude_modules": null,
|
| 12 |
+
"fan_in_fan_out": false,
|
| 13 |
+
"inference_mode": true,
|
| 14 |
+
"init_lora_weights": true,
|
| 15 |
+
"kasa_config": null,
|
| 16 |
+
"layer_replication": null,
|
| 17 |
+
"layers_pattern": null,
|
| 18 |
+
"layers_to_transform": null,
|
| 19 |
+
"loftq_config": {},
|
| 20 |
+
"lora_alpha": 32,
|
| 21 |
+
"lora_bias": false,
|
| 22 |
+
"lora_dropout": 0.05,
|
| 23 |
+
"lora_ga_config": null,
|
| 24 |
+
"megatron_config": null,
|
| 25 |
+
"megatron_core": "megatron.core",
|
| 26 |
+
"modules_to_save": null,
|
| 27 |
+
"monteclora_config": null,
|
| 28 |
+
"peft_type": "LORA",
|
| 29 |
+
"peft_version": "0.21.0",
|
| 30 |
+
"qalora_group_size": 16,
|
| 31 |
+
"r": 16,
|
| 32 |
+
"rank_pattern": {},
|
| 33 |
+
"revision": null,
|
| 34 |
+
"target_modules": [
|
| 35 |
+
"up_proj",
|
| 36 |
+
"k_proj",
|
| 37 |
+
"v_proj",
|
| 38 |
+
"gate_proj",
|
| 39 |
+
"q_proj",
|
| 40 |
+
"down_proj",
|
| 41 |
+
"o_proj"
|
| 42 |
+
],
|
| 43 |
+
"target_parameters": null,
|
| 44 |
+
"task_type": "CAUSAL_LM",
|
| 45 |
+
"trainable_token_indices": null,
|
| 46 |
+
"use_bdlora": null,
|
| 47 |
+
"use_dora": false,
|
| 48 |
+
"use_qalora": false,
|
| 49 |
+
"use_rslora": false,
|
| 50 |
+
"velora_config": null
|
| 51 |
+
}
|
06b/adapter_model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c77772b101ad6597ee368a0ba722e7751726cefdb46198ccc25b881a75e9b295
|
| 3 |
+
size 40422168
|
4b/adapter_config.json
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"alora_invocation_tokens": null,
|
| 3 |
+
"alpha_pattern": {},
|
| 4 |
+
"arrow_config": null,
|
| 5 |
+
"auto_mapping": null,
|
| 6 |
+
"base_model_name_or_path": "Qwen/Qwen3-4B-Instruct-2507",
|
| 7 |
+
"bias": "none",
|
| 8 |
+
"corda_config": null,
|
| 9 |
+
"ensure_weight_tying": false,
|
| 10 |
+
"eva_config": null,
|
| 11 |
+
"exclude_modules": null,
|
| 12 |
+
"fan_in_fan_out": false,
|
| 13 |
+
"inference_mode": true,
|
| 14 |
+
"init_lora_weights": true,
|
| 15 |
+
"kasa_config": null,
|
| 16 |
+
"layer_replication": null,
|
| 17 |
+
"layers_pattern": null,
|
| 18 |
+
"layers_to_transform": null,
|
| 19 |
+
"loftq_config": {},
|
| 20 |
+
"lora_alpha": 32,
|
| 21 |
+
"lora_bias": false,
|
| 22 |
+
"lora_dropout": 0.05,
|
| 23 |
+
"lora_ga_config": null,
|
| 24 |
+
"megatron_config": null,
|
| 25 |
+
"megatron_core": "megatron.core",
|
| 26 |
+
"modules_to_save": null,
|
| 27 |
+
"monteclora_config": null,
|
| 28 |
+
"peft_type": "LORA",
|
| 29 |
+
"peft_version": "0.21.0",
|
| 30 |
+
"qalora_group_size": 16,
|
| 31 |
+
"r": 16,
|
| 32 |
+
"rank_pattern": {},
|
| 33 |
+
"revision": null,
|
| 34 |
+
"target_modules": [
|
| 35 |
+
"o_proj",
|
| 36 |
+
"gate_proj",
|
| 37 |
+
"down_proj",
|
| 38 |
+
"k_proj",
|
| 39 |
+
"v_proj",
|
| 40 |
+
"up_proj",
|
| 41 |
+
"q_proj"
|
| 42 |
+
],
|
| 43 |
+
"target_parameters": null,
|
| 44 |
+
"task_type": "CAUSAL_LM",
|
| 45 |
+
"trainable_token_indices": null,
|
| 46 |
+
"use_bdlora": null,
|
| 47 |
+
"use_dora": false,
|
| 48 |
+
"use_qalora": false,
|
| 49 |
+
"use_rslora": false,
|
| 50 |
+
"velora_config": null
|
| 51 |
+
}
|
4b/adapter_model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:fb052fd249de8100f9898e2f4bec5f8300df5cd2f22ee6b45f3a95bebe93715f
|
| 3 |
+
size 132187888
|
README.md
ADDED
|
@@ -0,0 +1,83 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
---
|
| 2 |
+
license: apache-2.0
|
| 3 |
+
library_name: peft
|
| 4 |
+
tags:
|
| 5 |
+
- decision-model
|
| 6 |
+
- system-one
|
| 7 |
+
- calibration
|
| 8 |
+
- lora
|
| 9 |
+
- qlora
|
| 10 |
+
- typesafe
|
| 11 |
+
base_model:
|
| 12 |
+
- Qwen/Qwen3-4B-Instruct-2507
|
| 13 |
+
- Qwen/Qwen3-0.6B
|
| 14 |
+
metrics:
|
| 15 |
+
- accuracy
|
| 16 |
+
- brier_score
|
| 17 |
+
- expected_calibration_error
|
| 18 |
+
---
|
| 19 |
+
|
| 20 |
+
# System-One QLoRA (0.6B + 4B)
|
| 21 |
+
|
| 22 |
+
Open **single-pass decision scorers** in the shape of TypeSafe's *Jev* ("System One"):
|
| 23 |
+
a `state` + a multiple-choice `question` go in, a **calibrated distribution over the
|
| 24 |
+
options** comes out in one forward pass — no autoregressive generation.
|
| 25 |
+
|
| 26 |
+
Two QLoRA adapters, same recipe, two latency/accuracy points:
|
| 27 |
+
|
| 28 |
+
| | base | trainable | latency / answer | where it wins vs Jev |
|
| 29 |
+
|---|---|---|---|
|
| 30 |
+
| **`06b/`** | Qwen3-0.6B | LoRA r=16 | **~53 ms** (20× faster than Jev) | abstention 0.960 vs 0.740 |
|
| 31 |
+
| **`4b/`** | Qwen3-4B-Instruct-2507 | LoRA r=16 | **~165 ms** (6.5× faster) | reflex + abstention; best SNI |
|
| 32 |
+
|
| 33 |
+
## Scoring mechanism
|
| 34 |
+
|
| 35 |
+
Prompt ends `…\n\nAnswer:`; we read the next-token log-prob of each option **letter**
|
| 36 |
+
(` A`, ` B`, …), softmax over the options, and apply a temperature fitted on val.
|
| 37 |
+
The argmax is the decision; the softmax is the calibrated confidence. (Letter readout
|
| 38 |
+
caps at 26 options — costs 1 of ~3,350 benchmark rows here.)
|
| 39 |
+
|
| 40 |
+
```python
|
| 41 |
+
import torch
|
| 42 |
+
from peft import PeftModel
|
| 43 |
+
from transformers import AutoModelForCausalLM, AutoTokenizer
|
| 44 |
+
|
| 45 |
+
base = "Qwen/Qwen3-4B-Instruct-2507" # or "Qwen/Qwen3-0.6B" for 06b/
|
| 46 |
+
tok = AutoTokenizer.from_pretrained(base)
|
| 47 |
+
model = AutoModelForCausalLM.from_pretrained(base, torch_dtype=torch.bfloat16, device_map="auto")
|
| 48 |
+
model = PeftModel.from_pretrained(model, "yocoms/system1-qlora", subfolder="4b") # or "06b"
|
| 49 |
+
model.eval()
|
| 50 |
+
|
| 51 |
+
prompt = "State:\n{state}\n\nQuestion:\n{q}\n\nOptions:\nA. ...\nB. ...\n\nAnswer:"
|
| 52 |
+
ids = tok(prompt, return_tensors="pt").to(model.device)
|
| 53 |
+
logits = model(**ids).logits[0, -1]
|
| 54 |
+
# compare log-probs of ' A', ' B', ... ; softmax(/T) = calibrated distribution
|
| 55 |
+
```
|
| 56 |
+
|
| 57 |
+
## Benchmarks (held-out test, calibrated)
|
| 58 |
+
|
| 59 |
+
Accuracy / ECE, vs frozen **Jev 1.13** (`opencode-zen/jev-1.13`) on identical items.
|
| 60 |
+
|
| 61 |
+
| benchmark | 0.6B | 4B | Jev |
|
| 62 |
+
|---|---|---|---|
|
| 63 |
+
| SNI (semantic / NLI) | 0.613 / .037 | **0.707** / .050 | 0.838 |
|
| 64 |
+
| reflex (safety) | 0.553 / .059 | 0.558 / .072 | 0.543 |
|
| 65 |
+
| BFCL (tool select) | 0.885 / .032 | 0.920 / .034 | 0.957 |
|
| 66 |
+
| **abstention** (none-fits) | **0.960** / .019 | 0.924 / .035 | 0.740 |
|
| 67 |
+
|
| 68 |
+
- **Beats Jev** on reflex and abstention; near-parity on tool selection; well-calibrated (ECE ≤ 0.07) across the board.
|
| 69 |
+
- **SNI gap to Jev is knowledge/reasoning**, not format — see recipe.
|
| 70 |
+
|
| 71 |
+
## Recipe (why it works)
|
| 72 |
+
|
| 73 |
+
- Corpus: SNI + reflex train splits, **absence-augmentation** (abstain/NOTA variants,
|
| 74 |
+
upweighted absence class → the abstention win), and **format-matched NLI**.
|
| 75 |
+
- **The NLI fix**: SNI's held-out NLI tasks are *"pick which of 3 candidates is neutral"*,
|
| 76 |
+
not single-pair label classification. Training standard MNLI (`premise+hypothesis→label`)
|
| 77 |
+
left `mnli_neutral` at 0.16 (below chance). Rebuilding the data in the *select-of-3* format
|
| 78 |
+
lifted `mnli_neutral` **0.16 → 0.88** (beating Jev's 0.84) on the 0.6B — it was a format
|
| 79 |
+
mismatch, not model capacity.
|
| 80 |
+
- QLoRA 4-bit NF4, r=16, batch 4 × seq 384, ~1.25 epoch. Code + corpus builders:
|
| 81 |
+
https://github.com/y0c0ms/System1QLoRa
|
| 82 |
+
|
| 83 |
+
Not affiliated with TypeSafe AI. Independent reproduction of the *shape* of Jev.
|