apolo13x commited on
Commit
3051796
·
verified ·
1 Parent(s): 8a70f16

Add Nemotron 3.5 DFlash GGUF

Browse files
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ dflash-NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4.gguf filter=lfs diff=lfs merge=lfs -text
README.md ADDED
@@ -0,0 +1,41 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ license: other
3
+ license_name: openmdw-1.1
4
+ license_link: https://openmdw.ai/license/1-1/
5
+ base_model: nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4-DFlash
6
+ base_model_relation: quantized
7
+ tags:
8
+ - gguf
9
+ - dflash
10
+ - nvfp4
11
+ ---
12
+
13
+ # Nemotron 3.5 Lightning DFlash GGUF
14
+
15
+ This is the GGUF conversion of NVIDIA's [Nemotron 3.5 Lightning DFlash checkpoint](https://huggingface.co/nvidia/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4-DFlash).
16
+
17
+ It must be paired with the original model. For example:
18
+
19
+ ```bash
20
+ llama-server \
21
+ -hf ggml-org/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-GGUF:Q4_K_M \
22
+ -hfd apolo13x/NVIDIA-Nemotron-3.5-Lightning-30B-A3B-DFlash-GGUF \
23
+ --spec-type draft-dflash \
24
+ -ngl all \
25
+ -ngld all \
26
+ -fa on \
27
+ --temp 1.0 \
28
+ --top-p 0.95
29
+ ```
30
+
31
+ NVIDIA recommends temperature `1.0` and top-p `0.95`.
32
+
33
+ This DFlash GGUF was obtained with llama.cpp `b10373` by running:
34
+
35
+ ```bash
36
+ python3 convert_hf_to_gguf.py \
37
+ dflash-hf \
38
+ --target-model-dir target-meta \
39
+ --outtype bf16 \
40
+ --outfile dflash-NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4.gguf
41
+ ```
dflash-NVIDIA-Nemotron-3.5-Lightning-30B-A3B-NVFP4.gguf ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:60071ad805c56e6cc663c5d2391d0000d50df5415b1199f963bec55ecc6164f2
3
+ size 1185032768