File size: 1,040 Bytes
d79e9e7
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
{
  "schemaVersion": 1,
  "environment": {
    "CUDA_VISIBLE_DEVICES": ""
  },
  "quantize": [
    "<REPO_ROOT>/llama.cpp/build/bin/llama-quantize",
    "--allow-requantize",
    "--imatrix",
    "hf://unsloth/Qwen3.8-27B-GGUF/imatrix_unsloth.gguf",
    "--tensor-type-file",
    "<REPO_ROOT>/magicquant-manifest/experiments/UD-Q4_K_S-Unsloth/effective-tensor-types.txt",
    "<REPO_ROOT>/Qwen3.8-27B-Quark-AWQ-MXFP4-native.gguf",
    "<SCRATCH_HOME>/lane-home/Qwen3.8-27B-Quark-MXFP4-UD-Q4_K_S-Unsloth.gguf",
    "Q4_K_S",
    "16"
  ],
  "candidateKld": [
    "<REPO_ROOT>/llama.cpp/build/bin/llama-perplexity",
    "-m",
    "<SCRATCH_HOME>/lane-home/Qwen3.8-27B-Quark-MXFP4-UD-Q4_K_S-Unsloth.gguf",
    "-ngl",
    "0",
    "-t",
    "4",
    "-c",
    "2048",
    "--file",
    "<REPO_ROOT>/Experiments/MQ-IQ4_XS_1-Generic/benchmark/_ppl_corpora/ppl_corpus_general.txt",
    "--kl-divergence-base",
    "<REPO_ROOT>/Experiments/MQ-IQ4_XS_1-Generic/benchmark/native-reference/logits/kld_logits_general.bin",
    "--kl-divergence"
  ]
}