aoiandroid commited on
Commit
ccd8e9a
·
verified ·
1 Parent(s): 1c3cd91

Enable AutoModelForCausalLM direct loading with Quanto FP8 quantization_config

Browse files
Files changed (1) hide show
  1. config.json +7 -2
config.json CHANGED
@@ -29,5 +29,10 @@
29
  "tie_word_embeddings": false,
30
  "transformers_version": "4.57.6",
31
  "use_cache": true,
32
- "vocab_size": 130560
33
- }
 
 
 
 
 
 
29
  "tie_word_embeddings": false,
30
  "transformers_version": "4.57.6",
31
  "use_cache": true,
32
+ "vocab_size": 130560,
33
+ "quantization_config": {
34
+ "quant_method": "quanto",
35
+ "weights": "float8",
36
+ "activations": null
37
+ }
38
+ }