joseplcam commited on
Commit
c10a8f8
·
verified ·
1 Parent(s): 7cbf234

Inline kernel setup: Diffusers downloads only each component's own module

Browse files
NOTICE CHANGED
@@ -30,4 +30,4 @@ MODIFIED FILES (changed by joseplcam):
30
  The `text_encoder` and `transformer` entries point to the custom classes below.
31
 
32
  NEW FILES: text_encoder/modeling_nunchaku_qwen3vl.py, transformer/modeling_nunchaku_qwenimage21.py,
33
- text_encoder/nunchaku_kernels.py, transformer/nunchaku_kernels.py.
 
30
  The `text_encoder` and `transformer` entries point to the custom classes below.
31
 
32
  NEW FILES: text_encoder/modeling_nunchaku_qwen3vl.py, transformer/modeling_nunchaku_qwenimage21.py,
33
+ tools/ (the scripts that produced this repository).
README.md CHANGED
@@ -48,11 +48,13 @@ edited = pipe("Make it night time with moonlight", image=image, output_resolutio
48
 
49
  - `text_encoder/modeling_nunchaku_qwen3vl.py`: a `Qwen3VLForConditionalGeneration` subclass that swaps the
50
  quantized linears for Diffusers' own `SVDQW4A4Linear` before loading the weights.
51
- - `nunchaku_kernels.py` (in `text_encoder/` and `transformer/`): Diffusers loads the NVFP4 kernels from
52
- `rootonchair/nunchaku-lite-kernels`, which is no longer downloadable. This file points that name at
53
- [joseplcam/nunchaku-lite-kernels](https://huggingface.co/joseplcam/nunchaku-lite-kernels), an unmodified
54
- build of the same open-source kernels, and sets `DIFFUSERS_TRUST_REMOTE_KERNELS=true` unless you already set
55
- it. Set `LOCAL_KERNELS` yourself to use a different build.
 
 
56
 
57
  Use a different seed for an edit than the one that generated its input image. Qwen-Image-2.1 returns an
58
  over-sharpened copy that ignores the prompt when the edit starts from the same noise
 
48
 
49
  - `text_encoder/modeling_nunchaku_qwen3vl.py`: a `Qwen3VLForConditionalGeneration` subclass that swaps the
50
  quantized linears for Diffusers' own `SVDQW4A4Linear` before loading the weights.
51
+ - `transformer/modeling_nunchaku_qwenimage21.py`: the stock transformer class, unchanged.
52
+
53
+ Both files start with the same kernel setup, so it runs whichever component loads first. Diffusers loads
54
+ the NVFP4 kernels from `rootonchair/nunchaku-lite-kernels`, which is no longer downloadable. The setup
55
+ points that name at [joseplcam/nunchaku-lite-kernels](https://huggingface.co/joseplcam/nunchaku-lite-kernels),
56
+ an unmodified build of the same open-source kernels, and sets `DIFFUSERS_TRUST_REMOTE_KERNELS=true` unless
57
+ you already set it. Set `LOCAL_KERNELS` yourself to use a different build.
58
 
59
  Use a different seed for an edit than the one that generated its input image. Qwen-Image-2.1 returns an
60
  over-sharpened copy that ignores the prompt when the edit starts from the same noise
text_encoder/modeling_nunchaku_qwen3vl.py CHANGED
@@ -8,15 +8,31 @@ transformer. Everything else (vision tower, embeddings, the BF16 layers) loads
8
  as in the stock model.
9
  """
10
 
 
11
  from pathlib import Path
12
 
13
  import torch
14
  from accelerate import init_empty_weights
15
  from huggingface_hub import snapshot_download
16
 
17
- from .nunchaku_kernels import KERNELS # noqa: F401 (sets up the kernels before Diffusers imports them)
18
-
19
- # isort: off -- must stay below `.nunchaku_kernels`, which prepares the kernels this import loads
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
20
  from diffusers.quantizers.nunchaku.utils import replace_with_nunchaku_linear # noqa: E402
21
  from safetensors.torch import load_file # noqa: E402
22
  from transformers import Qwen3VLForConditionalGeneration # noqa: E402
 
8
  as in the stock model.
9
  """
10
 
11
+ import os
12
  from pathlib import Path
13
 
14
  import torch
15
  from accelerate import init_empty_weights
16
  from huggingface_hub import snapshot_download
17
 
18
+ # --- Kernel setup. Kept identical in transformer/modeling_nunchaku_qwenimage21.py: Diffusers downloads
19
+ # only each component's own module file, and whichever component loads first must run it. ---
20
+ # Diffusers fetches the Nunchaku Lite kernels from `rootonchair/nunchaku-lite-kernels`, which is
21
+ # no longer downloadable. Loading this repo with trust_remote_code=True already runs this code,
22
+ # so point that kernel name at the rebuilt copy before Diffusers imports its Nunchaku utilities.
23
+ KERNELS = "rootonchair/nunchaku-lite-kernels"
24
+ if KERNELS not in os.environ.get("LOCAL_KERNELS", ""):
25
+ local = f"{KERNELS}={snapshot_download('joseplcam/nunchaku-lite-kernels')}"
26
+ os.environ["LOCAL_KERNELS"] = ":".join(filter(None, [os.environ.get("LOCAL_KERNELS"), local]))
27
+ if "DIFFUSERS_TRUST_REMOTE_KERNELS" not in os.environ:
28
+ # Diffusers read this variable into a constant at import time, so set both.
29
+ import diffusers.utils.constants
30
+
31
+ os.environ["DIFFUSERS_TRUST_REMOTE_KERNELS"] = "true"
32
+ diffusers.utils.constants.DIFFUSERS_TRUST_REMOTE_KERNELS = True
33
+ # --- end of kernel setup ---
34
+
35
+ # isort: off -- must stay below the kernel setup, which prepares the kernels this import loads
36
  from diffusers.quantizers.nunchaku.utils import replace_with_nunchaku_linear # noqa: E402
37
  from safetensors.torch import load_file # noqa: E402
38
  from transformers import Qwen3VLForConditionalGeneration # noqa: E402
text_encoder/nunchaku_kernels.py DELETED
@@ -1,19 +0,0 @@
1
- # Shared by text_encoder/ and transformer/: whichever component Diffusers loads first sets this up.
2
- #
3
- # Diffusers fetches the Nunchaku Lite kernels from `rootonchair/nunchaku-lite-kernels`, which is
4
- # no longer downloadable. Loading this repo with trust_remote_code=True already runs this code,
5
- # so point that kernel name at the rebuilt copy before Diffusers imports its Nunchaku utilities.
6
- import os
7
-
8
- from huggingface_hub import snapshot_download
9
-
10
- KERNELS = "rootonchair/nunchaku-lite-kernels"
11
- if KERNELS not in os.environ.get("LOCAL_KERNELS", ""):
12
- local = f"{KERNELS}={snapshot_download('joseplcam/nunchaku-lite-kernels')}"
13
- os.environ["LOCAL_KERNELS"] = ":".join(filter(None, [os.environ.get("LOCAL_KERNELS"), local]))
14
- if "DIFFUSERS_TRUST_REMOTE_KERNELS" not in os.environ:
15
- # Diffusers read this variable into a constant at import time, so set both.
16
- import diffusers.utils.constants
17
-
18
- os.environ["DIFFUSERS_TRUST_REMOTE_KERNELS"] = "true"
19
- diffusers.utils.constants.DIFFUSERS_TRUST_REMOTE_KERNELS = True
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
tools/modeling_nunchaku_qwen3vl.py CHANGED
@@ -8,15 +8,31 @@ transformer. Everything else (vision tower, embeddings, the BF16 layers) loads
8
  as in the stock model.
9
  """
10
 
 
11
  from pathlib import Path
12
 
13
  import torch
14
  from accelerate import init_empty_weights
15
  from huggingface_hub import snapshot_download
16
 
17
- from .nunchaku_kernels import KERNELS # noqa: F401 (sets up the kernels before Diffusers imports them)
18
-
19
- # isort: off -- must stay below `.nunchaku_kernels`, which prepares the kernels this import loads
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
20
  from diffusers.quantizers.nunchaku.utils import replace_with_nunchaku_linear # noqa: E402
21
  from safetensors.torch import load_file # noqa: E402
22
  from transformers import Qwen3VLForConditionalGeneration # noqa: E402
 
8
  as in the stock model.
9
  """
10
 
11
+ import os
12
  from pathlib import Path
13
 
14
  import torch
15
  from accelerate import init_empty_weights
16
  from huggingface_hub import snapshot_download
17
 
18
+ # --- Kernel setup. Kept identical in transformer/modeling_nunchaku_qwenimage21.py: Diffusers downloads
19
+ # only each component's own module file, and whichever component loads first must run it. ---
20
+ # Diffusers fetches the Nunchaku Lite kernels from `rootonchair/nunchaku-lite-kernels`, which is
21
+ # no longer downloadable. Loading this repo with trust_remote_code=True already runs this code,
22
+ # so point that kernel name at the rebuilt copy before Diffusers imports its Nunchaku utilities.
23
+ KERNELS = "rootonchair/nunchaku-lite-kernels"
24
+ if KERNELS not in os.environ.get("LOCAL_KERNELS", ""):
25
+ local = f"{KERNELS}={snapshot_download('joseplcam/nunchaku-lite-kernels')}"
26
+ os.environ["LOCAL_KERNELS"] = ":".join(filter(None, [os.environ.get("LOCAL_KERNELS"), local]))
27
+ if "DIFFUSERS_TRUST_REMOTE_KERNELS" not in os.environ:
28
+ # Diffusers read this variable into a constant at import time, so set both.
29
+ import diffusers.utils.constants
30
+
31
+ os.environ["DIFFUSERS_TRUST_REMOTE_KERNELS"] = "true"
32
+ diffusers.utils.constants.DIFFUSERS_TRUST_REMOTE_KERNELS = True
33
+ # --- end of kernel setup ---
34
+
35
+ # isort: off -- must stay below the kernel setup, which prepares the kernels this import loads
36
  from diffusers.quantizers.nunchaku.utils import replace_with_nunchaku_linear # noqa: E402
37
  from safetensors.torch import load_file # noqa: E402
38
  from transformers import Qwen3VLForConditionalGeneration # noqa: E402
tools/modeling_nunchaku_qwenimage21.py CHANGED
@@ -1,13 +1,32 @@
1
  """Qwen-Image-2.1 transformer, unchanged, loaded as a custom component.
2
 
3
- Its only job is importing `nunchaku_kernels`, so the NVFP4 kernels are set up even when
4
  Diffusers loads this component before the text encoder. The Nunchaku Lite quantization
5
  itself is handled by Diffusers' built-in quantizer from `config.json`.
6
  """
7
 
8
- from diffusers import QwenImage21Transformer2DModel
9
 
10
- from .nunchaku_kernels import KERNELS # noqa: F401
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11
 
12
 
13
  class NunchakuQwenImage21Transformer2DModel(QwenImage21Transformer2DModel):
 
1
  """Qwen-Image-2.1 transformer, unchanged, loaded as a custom component.
2
 
3
+ Its only job is running the kernel setup below, so the NVFP4 kernels are ready even when
4
  Diffusers loads this component before the text encoder. The Nunchaku Lite quantization
5
  itself is handled by Diffusers' built-in quantizer from `config.json`.
6
  """
7
 
8
+ import os
9
 
10
+ from huggingface_hub import snapshot_download
11
+
12
+ # --- Kernel setup. Kept identical in text_encoder/modeling_nunchaku_qwen3vl.py: Diffusers downloads
13
+ # only each component's own module file, and whichever component loads first must run it. ---
14
+ # Diffusers fetches the Nunchaku Lite kernels from `rootonchair/nunchaku-lite-kernels`, which is
15
+ # no longer downloadable. Loading this repo with trust_remote_code=True already runs this code,
16
+ # so point that kernel name at the rebuilt copy before Diffusers imports its Nunchaku utilities.
17
+ KERNELS = "rootonchair/nunchaku-lite-kernels"
18
+ if KERNELS not in os.environ.get("LOCAL_KERNELS", ""):
19
+ local = f"{KERNELS}={snapshot_download('joseplcam/nunchaku-lite-kernels')}"
20
+ os.environ["LOCAL_KERNELS"] = ":".join(filter(None, [os.environ.get("LOCAL_KERNELS"), local]))
21
+ if "DIFFUSERS_TRUST_REMOTE_KERNELS" not in os.environ:
22
+ # Diffusers read this variable into a constant at import time, so set both.
23
+ import diffusers.utils.constants
24
+
25
+ os.environ["DIFFUSERS_TRUST_REMOTE_KERNELS"] = "true"
26
+ diffusers.utils.constants.DIFFUSERS_TRUST_REMOTE_KERNELS = True
27
+ # --- end of kernel setup ---
28
+
29
+ from diffusers import QwenImage21Transformer2DModel # noqa: E402
30
 
31
 
32
  class NunchakuQwenImage21Transformer2DModel(QwenImage21Transformer2DModel):
tools/nunchaku_kernels.py DELETED
@@ -1,19 +0,0 @@
1
- # Shared by text_encoder/ and transformer/: whichever component Diffusers loads first sets this up.
2
- #
3
- # Diffusers fetches the Nunchaku Lite kernels from `rootonchair/nunchaku-lite-kernels`, which is
4
- # no longer downloadable. Loading this repo with trust_remote_code=True already runs this code,
5
- # so point that kernel name at the rebuilt copy before Diffusers imports its Nunchaku utilities.
6
- import os
7
-
8
- from huggingface_hub import snapshot_download
9
-
10
- KERNELS = "rootonchair/nunchaku-lite-kernels"
11
- if KERNELS not in os.environ.get("LOCAL_KERNELS", ""):
12
- local = f"{KERNELS}={snapshot_download('joseplcam/nunchaku-lite-kernels')}"
13
- os.environ["LOCAL_KERNELS"] = ":".join(filter(None, [os.environ.get("LOCAL_KERNELS"), local]))
14
- if "DIFFUSERS_TRUST_REMOTE_KERNELS" not in os.environ:
15
- # Diffusers read this variable into a constant at import time, so set both.
16
- import diffusers.utils.constants
17
-
18
- os.environ["DIFFUSERS_TRUST_REMOTE_KERNELS"] = "true"
19
- diffusers.utils.constants.DIFFUSERS_TRUST_REMOTE_KERNELS = True
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
tools/package.py CHANGED
@@ -27,14 +27,24 @@ from examples.convert_nunchaku_lite_diffusers import ( # noqa: E402
27
  )
28
 
29
  HERE = Path(__file__).parent
30
- # component -> (module file, class); both import nunchaku_kernels, which ships next to each
31
  CUSTOM_COMPONENTS = {
32
  "text_encoder": ("modeling_nunchaku_qwen3vl", "NunchakuQwen3VLForConditionalGeneration"),
33
  "transformer": ("modeling_nunchaku_qwenimage21", "NunchakuQwenImage21Transformer2DModel"),
34
  }
35
 
36
 
 
 
 
 
 
 
37
  def main():
 
 
 
 
38
  parser = argparse.ArgumentParser()
39
  parser.add_argument("--dit", required=True)
40
  parser.add_argument("--text-encoder", required=True)
@@ -55,7 +65,6 @@ def main():
55
  index = json.loads((out / "model_index.json").read_text())
56
  for component, (module, cls) in CUSTOM_COMPONENTS.items():
57
  shutil.copy2(HERE / f"{module}.py", out / component / f"{module}.py")
58
- shutil.copy2(HERE / "nunchaku_kernels.py", out / component / "nunchaku_kernels.py")
59
  index[component] = [module, cls]
60
  (out / "model_index.json").write_text(json.dumps(index, indent=2) + "\n")
61
 
@@ -66,7 +75,7 @@ def main():
66
  vae_config.pop("_name_or_path", None) # a local path, meaningless on the Hub
67
  (out / "vae" / "config.json").write_text(json.dumps(vae_config, indent=2, sort_keys=True) + "\n")
68
 
69
- for leftover in ("assets", ".gitattributes"):
70
  path = out / leftover
71
  shutil.rmtree(path) if path.is_dir() else path.unlink(missing_ok=True)
72
  for doc in ("README.md", "NOTICE"): # model card and license notice
 
27
  )
28
 
29
  HERE = Path(__file__).parent
30
+ # component -> (module file, class); both modules carry the same kernel setup block
31
  CUSTOM_COMPONENTS = {
32
  "text_encoder": ("modeling_nunchaku_qwen3vl", "NunchakuQwen3VLForConditionalGeneration"),
33
  "transformer": ("modeling_nunchaku_qwenimage21", "NunchakuQwenImage21Transformer2DModel"),
34
  }
35
 
36
 
37
+ def kernel_setup_block(module: str) -> str:
38
+ text = (HERE / f"{module}.py").read_text()
39
+ body = text.split("# --- Kernel setup.", 1)[1].split("# --- end of kernel setup ---", 1)[0]
40
+ return body.split("---\n", 1)[1] # drop the header line, which names the other file
41
+
42
+
43
  def main():
44
+ blocks = {kernel_setup_block(module) for module, _ in CUSTOM_COMPONENTS.values()}
45
+ if len(blocks) != 1:
46
+ raise ValueError("The kernel setup blocks in the two modeling files differ; keep them identical")
47
+
48
  parser = argparse.ArgumentParser()
49
  parser.add_argument("--dit", required=True)
50
  parser.add_argument("--text-encoder", required=True)
 
65
  index = json.loads((out / "model_index.json").read_text())
66
  for component, (module, cls) in CUSTOM_COMPONENTS.items():
67
  shutil.copy2(HERE / f"{module}.py", out / component / f"{module}.py")
 
68
  index[component] = [module, cls]
69
  (out / "model_index.json").write_text(json.dumps(index, indent=2) + "\n")
70
 
 
75
  vae_config.pop("_name_or_path", None) # a local path, meaningless on the Hub
76
  (out / "vae" / "config.json").write_text(json.dumps(vae_config, indent=2, sort_keys=True) + "\n")
77
 
78
+ for leftover in ("assets", ".gitattributes", ".cache"): # .cache: snapshot_download bookkeeping
79
  path = out / leftover
80
  shutil.rmtree(path) if path.is_dir() else path.unlink(missing_ok=True)
81
  for doc in ("README.md", "NOTICE"): # model card and license notice
transformer/modeling_nunchaku_qwenimage21.py CHANGED
@@ -1,13 +1,32 @@
1
  """Qwen-Image-2.1 transformer, unchanged, loaded as a custom component.
2
 
3
- Its only job is importing `nunchaku_kernels`, so the NVFP4 kernels are set up even when
4
  Diffusers loads this component before the text encoder. The Nunchaku Lite quantization
5
  itself is handled by Diffusers' built-in quantizer from `config.json`.
6
  """
7
 
8
- from diffusers import QwenImage21Transformer2DModel
9
 
10
- from .nunchaku_kernels import KERNELS # noqa: F401
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
11
 
12
 
13
  class NunchakuQwenImage21Transformer2DModel(QwenImage21Transformer2DModel):
 
1
  """Qwen-Image-2.1 transformer, unchanged, loaded as a custom component.
2
 
3
+ Its only job is running the kernel setup below, so the NVFP4 kernels are ready even when
4
  Diffusers loads this component before the text encoder. The Nunchaku Lite quantization
5
  itself is handled by Diffusers' built-in quantizer from `config.json`.
6
  """
7
 
8
+ import os
9
 
10
+ from huggingface_hub import snapshot_download
11
+
12
+ # --- Kernel setup. Kept identical in text_encoder/modeling_nunchaku_qwen3vl.py: Diffusers downloads
13
+ # only each component's own module file, and whichever component loads first must run it. ---
14
+ # Diffusers fetches the Nunchaku Lite kernels from `rootonchair/nunchaku-lite-kernels`, which is
15
+ # no longer downloadable. Loading this repo with trust_remote_code=True already runs this code,
16
+ # so point that kernel name at the rebuilt copy before Diffusers imports its Nunchaku utilities.
17
+ KERNELS = "rootonchair/nunchaku-lite-kernels"
18
+ if KERNELS not in os.environ.get("LOCAL_KERNELS", ""):
19
+ local = f"{KERNELS}={snapshot_download('joseplcam/nunchaku-lite-kernels')}"
20
+ os.environ["LOCAL_KERNELS"] = ":".join(filter(None, [os.environ.get("LOCAL_KERNELS"), local]))
21
+ if "DIFFUSERS_TRUST_REMOTE_KERNELS" not in os.environ:
22
+ # Diffusers read this variable into a constant at import time, so set both.
23
+ import diffusers.utils.constants
24
+
25
+ os.environ["DIFFUSERS_TRUST_REMOTE_KERNELS"] = "true"
26
+ diffusers.utils.constants.DIFFUSERS_TRUST_REMOTE_KERNELS = True
27
+ # --- end of kernel setup ---
28
+
29
+ from diffusers import QwenImage21Transformer2DModel # noqa: E402
30
 
31
 
32
  class NunchakuQwenImage21Transformer2DModel(QwenImage21Transformer2DModel):
transformer/nunchaku_kernels.py DELETED
@@ -1,19 +0,0 @@
1
- # Shared by text_encoder/ and transformer/: whichever component Diffusers loads first sets this up.
2
- #
3
- # Diffusers fetches the Nunchaku Lite kernels from `rootonchair/nunchaku-lite-kernels`, which is
4
- # no longer downloadable. Loading this repo with trust_remote_code=True already runs this code,
5
- # so point that kernel name at the rebuilt copy before Diffusers imports its Nunchaku utilities.
6
- import os
7
-
8
- from huggingface_hub import snapshot_download
9
-
10
- KERNELS = "rootonchair/nunchaku-lite-kernels"
11
- if KERNELS not in os.environ.get("LOCAL_KERNELS", ""):
12
- local = f"{KERNELS}={snapshot_download('joseplcam/nunchaku-lite-kernels')}"
13
- os.environ["LOCAL_KERNELS"] = ":".join(filter(None, [os.environ.get("LOCAL_KERNELS"), local]))
14
- if "DIFFUSERS_TRUST_REMOTE_KERNELS" not in os.environ:
15
- # Diffusers read this variable into a constant at import time, so set both.
16
- import diffusers.utils.constants
17
-
18
- os.environ["DIFFUSERS_TRUST_REMOTE_KERNELS"] = "true"
19
- diffusers.utils.constants.DIFFUSERS_TRUST_REMOTE_KERNELS = True