Enable AutoModelForCausalLM direct loading with Quanto FP8 quantization_config
Browse files- config.json +7 -2
config.json
CHANGED
|
@@ -29,5 +29,10 @@
|
|
| 29 |
"tie_word_embeddings": false,
|
| 30 |
"transformers_version": "4.57.6",
|
| 31 |
"use_cache": true,
|
| 32 |
-
"vocab_size": 130560
|
| 33 |
-
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 29 |
"tie_word_embeddings": false,
|
| 30 |
"transformers_version": "4.57.6",
|
| 31 |
"use_cache": true,
|
| 32 |
+
"vocab_size": 130560,
|
| 33 |
+
"quantization_config": {
|
| 34 |
+
"quant_method": "quanto",
|
| 35 |
+
"weights": "float8",
|
| 36 |
+
"activations": null
|
| 37 |
+
}
|
| 38 |
+
}
|