File size: 1,223 Bytes
ca03306
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
{
    "comp_cgenerate_active": false,
    "comp_ctranslate_active": false,
    "comp_cwhisper_active": false,
    "comp_diffusers2_active": false,
    "comp_flux_caching_active": false,
    "comp_ifw_active": false,
    "comp_ipex_llm_active": false,
    "comp_onediff_active": false,
    "comp_step_caching_active": false,
    "comp_torch_compile_active": false,
    "comp_ws2t_active": false,
    "comp_x-fast_active": false,
    "prune_torch-structured_active": false,
    "prune_torch-unstructured_active": false,
    "quant_aqlm_active": false,
    "quant_awq_active": false,
    "quant_gptq_active": false,
    "quant_half_active": false,
    "quant_higgs_active": true,
    "quant_hqq_active": false,
    "quant_llm-int8_active": false,
    "quant_quanto_active": false,
    "quant_torch_dynamic_active": false,
    "quant_torch_static_active": false,
    "quant_higgs_group_size": 256,
    "quant_higgs_hadamard_size": 1024,
    "quant_higgs_modules_to_not_convert": "lm_head",
    "quant_higgs_p": 2,
    "quant_higgs_weight_bits": 2,
    "max_batch_size": 1,
    "device": "cuda",
    "cache_dir": "/home/ubuntu/.cache/pruna/tmpc6y5r_fu",
    "task": "",
    "save_load_fn": "higgs",
    "save_load_fn_args": {}
}