LJTSG commited on
Commit
7d5a851
·
verified ·
1 Parent(s): 5337847

NPU-Forge v0.1: permanent custom-model registry for FastFlowLM on Ryzen AI NPUs

Browse files
Files changed (4) hide show
  1. README.md +94 -0
  2. bin/register.js +45 -0
  3. register-admin.bat +11 -0
  4. registry.example.json +30 -0
README.md ADDED
@@ -0,0 +1,94 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: mit
3
+ tags:
4
+ - amd
5
+ - ryzen-ai
6
+ - npu
7
+ - xdna2
8
+ - fastflowlm
9
+ - strix-halo
10
+ - tools
11
+ pipeline_tag: text-generation
12
+ ---
13
+
14
+ # NPU-Forge 🔥 — keep your custom models alive on AMD Ryzen AI NPUs
15
+
16
+ **The problem nobody tells you about:** you convert a fine-tuned model to
17
+ [FastFlowLM](https://github.com/FastFlowLM/FastFlowLM)'s Q4NX format, register
18
+ it in `model_list.json`, it runs beautifully on your NPU… and then an FLM
19
+ update ships and your custom model **silently disappears**. The model files
20
+ survive on disk — but `model_list.json` lives in `C:\Program Files\flm\` and
21
+ every update resets it, de-registering every custom entry you ever added.
22
+
23
+ We hit this on a Strix Halo box: a fully converted Llama-3 fine-tune
24
+ (Stheno 8B), complete with compiled `.xclbin` NPU kernels, sitting in
25
+ `Documents\flm\models\` — invisible to `flm list` because an update had
26
+ eaten its registration.
27
+
28
+ **NPU-Forge fixes it permanently:**
29
+
30
+ - **`registry.json`** — YOUR registry, in user space, owned by you. Every
31
+ custom model entry lives here forever. FLM updates can't touch it.
32
+ - **`bin/register.js`** — idempotently merges your registry into FLM's
33
+ `model_list.json`. Run it any time a model "disappears" — takes one second,
34
+ backs up FLM's list (timestamped) before every write.
35
+ - **`register-admin.bat`** — double-click version that self-elevates (the
36
+ one part that needs admin is writing into Program Files).
37
+
38
+ ## Usage
39
+
40
+ 1. Convert your fine-tune to Q4NX with the official
41
+ [FLM_Q4NX_Converter](https://github.com/FastFlowLM/FLM_Q4NX_Converter)
42
+ (GGUF in — any quant — Q4NX out). Supported families: LLaMA, Qwen 2/2.5/3/3.5,
43
+ Gemma 3, Phi-4, LFM2, GPT-OSS, and more.
44
+ 2. Put the model folder in `C:\Users\<you>\Documents\flm\models\<YourModel>\`
45
+ with the standard four files: `config.json`, `model.q4nx`,
46
+ `tokenizer.json`, `tokenizer_config.json`. (No admin needed — model
47
+ storage is user-space.)
48
+ 3. Add an entry to `registry.json` (copy `registry.example.json` and edit —
49
+ easiest is mirroring the entry of the same-architecture stock model from
50
+ FLM's own `model_list.json`, changing `name` to your folder name and
51
+ blanking `url`).
52
+ 4. Double-click `register-admin.bat`. Done: `flm run yourmodel-forge:8b`.
53
+ 5. FLM updated and your model vanished again? Double-click again. That's it.
54
+
55
+ ```json
56
+ // registry.example.json — one custom llama3-family 8B model
57
+ {
58
+ "models": {
59
+ "mymodel-forge": {
60
+ "8b": {
61
+ "name": "MyModel-8B-NPU2",
62
+ "url": "",
63
+ "file_url": "",
64
+ "default_context_length": 8192,
65
+ "max_prefill_len": 4096,
66
+ "files": ["config.json", "model.q4nx", "tokenizer.json", "tokenizer_config.json"],
67
+ "details": { "family": "llama3", "parameter_size": "8B", "quantization_level": "Q4_1" },
68
+ "footprint": 4.7
69
+ }
70
+ }
71
+ }
72
+ }
73
+ ```
74
+
75
+ ## Facts we verified the hard way (Strix Halo, FLM v0.9.43)
76
+
77
+ - Model **files** are user-space (`Documents\flm\models\`) — no admin needed.
78
+ - The **registry** (`model_list.json`) is Program Files — admin only, and
79
+ reset by updates.
80
+ - A user-level `Documents\flm\model_list.json` is **ignored** (we tested).
81
+ - Unregistered model folders can't be run by folder name (we tested) —
82
+ registration is mandatory.
83
+ - FLM compiles per-model NPU kernels (`.xclbin`) on first run; they live in
84
+ the model folder and survive alongside it.
85
+
86
+ ## Where this is going
87
+
88
+ NPU-Forge is the first piece of a larger goal: **one-command fine-tune → NPU**.
89
+ The full pipeline (LoRA fine-tune in the cloud → GGUF → Q4NX → registered and
90
+ serving on your NPU, single command) is in active development, along with
91
+ encoder-model (BERT-class) NPU deployment. Watch this repo.
92
+
93
+ Built on a Ryzen AI MAX+ 395 (Strix Halo, XDNA2) as part of an ongoing
94
+ project to make local NPUs a first-class home for personal AI.
bin/register.js ADDED
@@ -0,0 +1,45 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ #!/usr/bin/env node
2
+ // register.js — merge npu-forge's own registry into FLM's model_list.json.
3
+ // Forge entries live in npu-forge\registry.json (user-space, permanent);
4
+ // FLM's list lives in Program Files and gets RESET by flm updates — so this
5
+ // merge is idempotent and re-runnable any time a custom model "disappears."
6
+ // Needs elevation (run via register-admin.bat). Always backs up first.
7
+ 'use strict';
8
+ const fs = require('fs');
9
+ const path = require('path');
10
+
11
+ const FORGE = path.join(__dirname, '..');
12
+ const FLM_LIST = 'C:\\Program Files\\flm\\model_list.json';
13
+ const REGISTRY = path.join(FORGE, 'registry.json');
14
+
15
+ function main() {
16
+ if (!fs.existsSync(REGISTRY)) { console.log('No registry.json — nothing to register.'); return; }
17
+ const reg = JSON.parse(fs.readFileSync(REGISTRY, 'utf8').replace(/^/, ''));
18
+ const ml = JSON.parse(fs.readFileSync(FLM_LIST, 'utf8').replace(/^/, ''));
19
+
20
+ // timestamped backup beside the forge registry (user-space, survives anything)
21
+ const stamp = new Date().toISOString().replace(/[:.]/g, '-');
22
+ fs.copyFileSync(FLM_LIST, path.join(FORGE, `model_list.backup-${stamp}.json`));
23
+
24
+ let added = 0, kept = 0;
25
+ for (const [key, sizes] of Object.entries(reg.models || {})) {
26
+ for (const [size, entry] of Object.entries(sizes)) {
27
+ ml.models[key] = ml.models[key] || {};
28
+ if (ml.models[key][size] && JSON.stringify(ml.models[key][size]) === JSON.stringify(entry)) { kept++; continue; }
29
+ ml.models[key][size] = entry;
30
+ added++;
31
+ console.log(`registered: ${key}:${size} -> ${entry.name}`);
32
+ }
33
+ }
34
+ try {
35
+ fs.writeFileSync(FLM_LIST, JSON.stringify(ml, null, 2));
36
+ } catch (e) {
37
+ console.error('\n[!] Could not write ' + FLM_LIST);
38
+ console.error(' This needs administrator rights — double-click register-admin.bat instead.');
39
+ process.exit(1);
40
+ }
41
+ console.log(`\nDone: ${added} entries written, ${kept} already current.`);
42
+ console.log('Verify with: flm list (your models end in -forge)');
43
+ }
44
+
45
+ main();
register-admin.bat ADDED
@@ -0,0 +1,11 @@
 
 
 
 
 
 
 
 
 
 
 
 
1
+ @echo off
2
+ title NPU-Forge - register custom models (admin)
3
+ net session >nul 2>&1
4
+ if errorlevel 1 (
5
+ echo Requesting administrator rights...
6
+ powershell -Command "Start-Process cmd -ArgumentList '/k cd /d \"%~dp0\" ^&^& node bin\register.js' -Verb RunAs"
7
+ exit /b
8
+ )
9
+ cd /d "%~dp0"
10
+ node "%~dp0bin\register.js"
11
+ pause
registry.example.json ADDED
@@ -0,0 +1,30 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "models": {
3
+ "mymodel-forge": {
4
+ "8b": {
5
+ "name": "MyModel-8B-NPU2",
6
+ "url": "",
7
+ "file_url": "",
8
+ "modified_at": "2026-06-10T00:00:00Z",
9
+ "size": 4700000000,
10
+ "flm_min_version": "0.9.38",
11
+ "default_context_length": 8192,
12
+ "max_prefill_len": 4096,
13
+ "files": [
14
+ "config.json",
15
+ "model.q4nx",
16
+ "tokenizer.json",
17
+ "tokenizer_config.json"
18
+ ],
19
+ "details": {
20
+ "family": "llama3",
21
+ "think": false,
22
+ "parameter_size": "8B",
23
+ "quantization_level": "Q4_1"
24
+ },
25
+ "label": [],
26
+ "footprint": 4.7
27
+ }
28
+ }
29
+ }
30
+ }