Instructions to use GreenBitAI/DeepSeek-R1-0528-Qwen3-8B-layer-mix-bpw-3.8-mlx with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- MLX
How to use GreenBitAI/DeepSeek-R1-0528-Qwen3-8B-layer-mix-bpw-3.8-mlx with MLX:
# Download the model from the Hub pip install huggingface_hub[hf_xet] huggingface-cli download --local-dir DeepSeek-R1-0528-Qwen3-8B-layer-mix-bpw-3.8-mlx GreenBitAI/DeepSeek-R1-0528-Qwen3-8B-layer-mix-bpw-3.8-mlx
- Notebooks
- Google Colab
- Kaggle
- Local Apps
- LM Studio
| { | |
| "measurement": { | |
| "model.layers.0": { | |
| "accuracy": 0.931088128243573, | |
| "total_bits": 593362944.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.1": { | |
| "accuracy": 0.9742336794734001, | |
| "total_bits": 832045056.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.2": { | |
| "accuracy": 0.9743453479022719, | |
| "total_bits": 832045056.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.3": { | |
| "accuracy": 0.9728944995440543, | |
| "total_bits": 832045056.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.4": { | |
| "accuracy": 0.9719419548637234, | |
| "total_bits": 832045056.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.5": { | |
| "accuracy": 0.9724181958008558, | |
| "total_bits": 832045056.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.6": { | |
| "accuracy": 0.9673449019319378, | |
| "total_bits": 832045056.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.7": { | |
| "accuracy": 0.9880880096025066, | |
| "total_bits": 786825216.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.8": { | |
| "accuracy": 0.9671264597272966, | |
| "total_bits": 593362944.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.9": { | |
| "accuracy": 0.9744890105794184, | |
| "total_bits": 682033152.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.10": { | |
| "accuracy": 0.9625228689983487, | |
| "total_bits": 593362944.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.11": { | |
| "accuracy": 0.9643016764312051, | |
| "total_bits": 593362944.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.12": { | |
| "accuracy": 0.975022288504988, | |
| "total_bits": 682033152.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.13": { | |
| "accuracy": 0.9616155670955777, | |
| "total_bits": 593362944.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.14": { | |
| "accuracy": 0.9737993401940912, | |
| "total_bits": 682033152.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.15": { | |
| "accuracy": 0.9623663239763118, | |
| "total_bits": 593362944.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.16": { | |
| "accuracy": 0.9750147936283611, | |
| "total_bits": 786825216.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.17": { | |
| "accuracy": 0.9721133908024058, | |
| "total_bits": 593362944.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.18": { | |
| "accuracy": 0.9761279110098258, | |
| "total_bits": 682033152.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.19": { | |
| "accuracy": 0.9639980566571467, | |
| "total_bits": 593362944.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.20": { | |
| "accuracy": 0.9638224755763076, | |
| "total_bits": 593362944.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.21": { | |
| "accuracy": 0.9833653933892492, | |
| "total_bits": 786825216.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.22": { | |
| "accuracy": 0.9810311227629427, | |
| "total_bits": 786825216.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.23": { | |
| "accuracy": 0.9849393175099976, | |
| "total_bits": 832045056.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.24": { | |
| "accuracy": 0.9839110090688337, | |
| "total_bits": 832045056.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.25": { | |
| "accuracy": 0.9803282860666513, | |
| "total_bits": 786825216.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.26": { | |
| "accuracy": 0.9809935875528026, | |
| "total_bits": 786825216.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.27": { | |
| "accuracy": 0.9793906190898269, | |
| "total_bits": 786825216.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.28": { | |
| "accuracy": 0.9828384715074208, | |
| "total_bits": 832045056.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.29": { | |
| "accuracy": 0.983858243591385, | |
| "total_bits": 832045056.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.30": { | |
| "accuracy": 0.9823532946174964, | |
| "total_bits": 832045056.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.31": { | |
| "accuracy": 0.9828728725260589, | |
| "total_bits": 832045056.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.32": { | |
| "accuracy": 0.9822730309679173, | |
| "total_bits": 832045056.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 32 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.33": { | |
| "accuracy": 0.9848659632261842, | |
| "total_bits": 786825216.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.34": { | |
| "accuracy": 0.9950092360377312, | |
| "total_bits": 593362944.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| }, | |
| "model.layers.35": { | |
| "accuracy": 0.9938810579478741, | |
| "total_bits": 593362944.0, | |
| "o_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "down_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "q_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "k_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "v_proj": { | |
| "group_size": { | |
| "4": 128 | |
| }, | |
| "bits": [ | |
| 4 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "gate_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| }, | |
| "up_proj": { | |
| "group_size": { | |
| "2": 64 | |
| }, | |
| "bits": [ | |
| 2 | |
| ], | |
| "bits_prop": [ | |
| 1 | |
| ], | |
| "scale_bits": 4, | |
| "scale_groups:": 32 | |
| } | |
| } | |
| } | |
| } |