diff --git "a/quant_strategy.json" "b/quant_strategy.json" new file mode 100644--- /dev/null +++ "b/quant_strategy.json" @@ -0,0 +1,5284 @@ +{ + "measurement": { + "model.layers.0": { + "accuracy": 0.9743846654891968, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.1": { + "accuracy": 0.9799767732620239, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.2": { + "accuracy": 0.931053876876831, + "total_bits": 1844214912, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.3": { + "accuracy": 0.992832601070404, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.4": { + "accuracy": 0.9924249649047852, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.5": { + "accuracy": 0.9917068779468536, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.6": { + "accuracy": 0.98964524269104, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.7": { + "accuracy": 0.9889870285987854, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.8": { + "accuracy": 0.9848525524139404, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.9": { + "accuracy": 0.9857602119445801, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.10": { + "accuracy": 0.9833030104637146, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.11": { + "accuracy": 0.9808352589607239, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.12": { + "accuracy": 0.9802364706993103, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.13": { + "accuracy": 0.9768545031547546, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.14": { + "accuracy": 0.975557804107666, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.15": { + "accuracy": 0.9787958860397339, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.16": { + "accuracy": 0.9775415658950806, + "total_bits": 1179493632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.17": { + "accuracy": 0.9723058938980103, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.18": { + "accuracy": 0.9702450037002563, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.19": { + "accuracy": 0.9673857688903809, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.20": { + "accuracy": 0.9692351818084717, + "total_bits": 1454051712, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.21": { + "accuracy": 0.9670658111572266, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.22": { + "accuracy": 0.9542118310928345, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.23": { + "accuracy": 0.9660637378692627, + "total_bits": 1179493632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.24": { + "accuracy": 0.9604368209838867, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.25": { + "accuracy": 0.9550518989562988, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.26": { + "accuracy": 0.9570149183273315, + "total_bits": 1179493632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.27": { + "accuracy": 0.9512978792190552, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.28": { + "accuracy": 0.9525103569030762, + "total_bits": 1179493632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.29": { + "accuracy": 0.950129508972168, + "total_bits": 1179493632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.30": { + "accuracy": 0.9424762725830078, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.31": { + "accuracy": 0.9367372989654541, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.32": { + "accuracy": 0.9346246719360352, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.33": { + "accuracy": 0.9341306686401367, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.34": { + "accuracy": 0.9288473129272461, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.35": { + "accuracy": 0.9245133399963379, + "total_bits": 1179493632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.36": { + "accuracy": 0.9215571880340576, + "total_bits": 1179493632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.37": { + "accuracy": 0.9287619590759277, + "total_bits": 1454051712, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.38": { + "accuracy": 0.9251844882965088, + "total_bits": 1454051712, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.39": { + "accuracy": 0.9258096218109131, + "total_bits": 1454051712, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.40": { + "accuracy": 0.9295849800109863, + "total_bits": 1454051712, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.41": { + "accuracy": 0.9270470142364502, + "total_bits": 1454051712, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.42": { + "accuracy": 0.9324800968170166, + "total_bits": 1454051712, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.43": { + "accuracy": 0.9354493618011475, + "total_bits": 1454051712, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.44": { + "accuracy": 0.9381983280181885, + "total_bits": 1454051712, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.45": { + "accuracy": 0.9263179302215576, + "total_bits": 1179493632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.46": { + "accuracy": 0.9267194271087646, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.47": { + "accuracy": 0.9308862686157227, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.48": { + "accuracy": 0.9323692321777344, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.49": { + "accuracy": 0.9348795413970947, + "total_bits": 1179493632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.50": { + "accuracy": 0.9369521141052246, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.51": { + "accuracy": 0.9379558563232422, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.52": { + "accuracy": 0.9416360855102539, + "total_bits": 1179493632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.53": { + "accuracy": 0.9430384635925293, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.54": { + "accuracy": 0.9454588890075684, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.55": { + "accuracy": 0.9455626010894775, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.56": { + "accuracy": 0.9423973560333252, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.57": { + "accuracy": 0.9447505474090576, + "total_bits": 1179493632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.58": { + "accuracy": 0.9341812133789062, + "total_bits": 1165045632, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + }, + "model.layers.59": { + "accuracy": 0.9366586208343506, + "total_bits": 1454051712, + "q_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "k_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "v_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "o_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "up_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "gate_proj": { + "group_size": { + "2": 64 + }, + "bits": [ + 2 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + }, + "down_proj": { + "group_size": { + "4": 128 + }, + "bits": [ + 4 + ], + "bits_prop": [ + 1 + ], + "scale_bits": 4 + } + } + } +} \ No newline at end of file