{ "quant_method": "exl3", "version": "1.5.0", "bits": 4.0, "head_bits": 6, "calibration": { "rows": 250, "cols": 2048 }, "out_scales": "always", "codebook": "mul1", "tensor_storage": { "fc": { "stored_tensors": { "fc.suh": { "shape": [ 25600 ], "n_bytes": 51200, "dtype": "torch.float16" }, "fc.svh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "fc.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "fc.trellis": { "shape": [ 1600, 320, 64 ], "n_bytes": 65536000, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "hidden_norm": { "stored_tensors": { "hidden_norm.weight": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.bfloat16" } } }, "layers.0.input_layernorm": { "stored_tensors": { "layers.0.input_layernorm.weight": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.bfloat16" } } }, "layers.0.self_attn.q_proj": { "stored_tensors": { "layers.0.self_attn.q_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.0.self_attn.q_proj.svh": { "shape": [ 4096 ], "n_bytes": 8192, "dtype": "torch.float16" }, "layers.0.self_attn.q_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.0.self_attn.q_proj.trellis": { "shape": [ 320, 256, 64 ], "n_bytes": 10485760, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.0.self_attn.k_proj": { "stored_tensors": { "layers.0.self_attn.k_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.0.self_attn.k_proj.svh": { "shape": [ 1024 ], "n_bytes": 2048, "dtype": "torch.float16" }, "layers.0.self_attn.k_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.0.self_attn.k_proj.trellis": { "shape": [ 320, 64, 64 ], "n_bytes": 2621440, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.0.self_attn.v_proj": { "stored_tensors": { "layers.0.self_attn.v_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.0.self_attn.v_proj.svh": { "shape": [ 1024 ], "n_bytes": 2048, "dtype": "torch.float16" }, "layers.0.self_attn.v_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.0.self_attn.v_proj.trellis": { "shape": [ 320, 64, 64 ], "n_bytes": 2621440, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.0.self_attn.o_proj": { "stored_tensors": { "layers.0.self_attn.o_proj.suh": { "shape": [ 4096 ], "n_bytes": 8192, "dtype": "torch.float16" }, "layers.0.self_attn.o_proj.svh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.0.self_attn.o_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.0.self_attn.o_proj.trellis": { "shape": [ 256, 320, 64 ], "n_bytes": 10485760, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.0.self_attn.q_norm": { "stored_tensors": { "layers.0.self_attn.q_norm.weight": { "shape": [ 128 ], "n_bytes": 256, "dtype": "torch.bfloat16" } } }, "layers.0.self_attn.k_norm": { "stored_tensors": { "layers.0.self_attn.k_norm.weight": { "shape": [ 128 ], "n_bytes": 256, "dtype": "torch.bfloat16" } } }, "layers.0.post_attention_layernorm": { "stored_tensors": { "layers.0.post_attention_layernorm.weight": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.bfloat16" } } }, "layers.0.mlp.up_proj": { "stored_tensors": { "layers.0.mlp.up_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.0.mlp.up_proj.svh": { "shape": [ 17408 ], "n_bytes": 34816, "dtype": "torch.float16" }, "layers.0.mlp.up_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.0.mlp.up_proj.trellis": { "shape": [ 320, 1088, 64 ], "n_bytes": 44564480, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.0.mlp.gate_proj": { "stored_tensors": { "layers.0.mlp.gate_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.0.mlp.gate_proj.svh": { "shape": [ 17408 ], "n_bytes": 34816, "dtype": "torch.float16" }, "layers.0.mlp.gate_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.0.mlp.gate_proj.trellis": { "shape": [ 320, 1088, 64 ], "n_bytes": 44564480, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.0.mlp.down_proj": { "stored_tensors": { "layers.0.mlp.down_proj.suh": { "shape": [ 17408 ], "n_bytes": 34816, "dtype": "torch.float16" }, "layers.0.mlp.down_proj.svh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.0.mlp.down_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.0.mlp.down_proj.trellis": { "shape": [ 1088, 320, 64 ], "n_bytes": 44564480, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.0.attention_conv.kernel_projection": { "stored_tensors": { "layers.0.attention_conv.kernel_projection.weight": { "shape": [ 1280, 5120 ], "n_bytes": 13107200, "dtype": "torch.float16" } } }, "layers.0.mlp_conv.kernel_projection": { "stored_tensors": { "layers.0.mlp_conv.kernel_projection.weight": { "shape": [ 1280, 5120 ], "n_bytes": 13107200, "dtype": "torch.float16" } } }, "layers.1.input_layernorm": { "stored_tensors": { "layers.1.input_layernorm.weight": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.bfloat16" } } }, "layers.1.self_attn.q_proj": { "stored_tensors": { "layers.1.self_attn.q_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.1.self_attn.q_proj.svh": { "shape": [ 4096 ], "n_bytes": 8192, "dtype": "torch.float16" }, "layers.1.self_attn.q_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.1.self_attn.q_proj.trellis": { "shape": [ 320, 256, 64 ], "n_bytes": 10485760, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.1.self_attn.k_proj": { "stored_tensors": { "layers.1.self_attn.k_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.1.self_attn.k_proj.svh": { "shape": [ 1024 ], "n_bytes": 2048, "dtype": "torch.float16" }, "layers.1.self_attn.k_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.1.self_attn.k_proj.trellis": { "shape": [ 320, 64, 64 ], "n_bytes": 2621440, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.1.self_attn.v_proj": { "stored_tensors": { "layers.1.self_attn.v_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.1.self_attn.v_proj.svh": { "shape": [ 1024 ], "n_bytes": 2048, "dtype": "torch.float16" }, "layers.1.self_attn.v_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.1.self_attn.v_proj.trellis": { "shape": [ 320, 64, 64 ], "n_bytes": 2621440, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.1.self_attn.o_proj": { "stored_tensors": { "layers.1.self_attn.o_proj.suh": { "shape": [ 4096 ], "n_bytes": 8192, "dtype": "torch.float16" }, "layers.1.self_attn.o_proj.svh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.1.self_attn.o_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.1.self_attn.o_proj.trellis": { "shape": [ 256, 320, 64 ], "n_bytes": 10485760, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.1.self_attn.q_norm": { "stored_tensors": { "layers.1.self_attn.q_norm.weight": { "shape": [ 128 ], "n_bytes": 256, "dtype": "torch.bfloat16" } } }, "layers.1.self_attn.k_norm": { "stored_tensors": { "layers.1.self_attn.k_norm.weight": { "shape": [ 128 ], "n_bytes": 256, "dtype": "torch.bfloat16" } } }, "layers.1.post_attention_layernorm": { "stored_tensors": { "layers.1.post_attention_layernorm.weight": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.bfloat16" } } }, "layers.1.mlp.up_proj": { "stored_tensors": { "layers.1.mlp.up_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.1.mlp.up_proj.svh": { "shape": [ 17408 ], "n_bytes": 34816, "dtype": "torch.float16" }, "layers.1.mlp.up_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.1.mlp.up_proj.trellis": { "shape": [ 320, 1088, 64 ], "n_bytes": 44564480, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.1.mlp.gate_proj": { "stored_tensors": { "layers.1.mlp.gate_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.1.mlp.gate_proj.svh": { "shape": [ 17408 ], "n_bytes": 34816, "dtype": "torch.float16" }, "layers.1.mlp.gate_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.1.mlp.gate_proj.trellis": { "shape": [ 320, 1088, 64 ], "n_bytes": 44564480, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.1.mlp.down_proj": { "stored_tensors": { "layers.1.mlp.down_proj.suh": { "shape": [ 17408 ], "n_bytes": 34816, "dtype": "torch.float16" }, "layers.1.mlp.down_proj.svh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.1.mlp.down_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.1.mlp.down_proj.trellis": { "shape": [ 1088, 320, 64 ], "n_bytes": 44564480, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.1.attention_conv.kernel_projection": { "stored_tensors": { "layers.1.attention_conv.kernel_projection.weight": { "shape": [ 1280, 5120 ], "n_bytes": 13107200, "dtype": "torch.float16" } } }, "layers.1.mlp_conv.kernel_projection": { "stored_tensors": { "layers.1.mlp_conv.kernel_projection.weight": { "shape": [ 1280, 5120 ], "n_bytes": 13107200, "dtype": "torch.float16" } } }, "layers.2.input_layernorm": { "stored_tensors": { "layers.2.input_layernorm.weight": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.bfloat16" } } }, "layers.2.self_attn.q_proj": { "stored_tensors": { "layers.2.self_attn.q_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.2.self_attn.q_proj.svh": { "shape": [ 4096 ], "n_bytes": 8192, "dtype": "torch.float16" }, "layers.2.self_attn.q_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.2.self_attn.q_proj.trellis": { "shape": [ 320, 256, 64 ], "n_bytes": 10485760, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.2.self_attn.k_proj": { "stored_tensors": { "layers.2.self_attn.k_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.2.self_attn.k_proj.svh": { "shape": [ 1024 ], "n_bytes": 2048, "dtype": "torch.float16" }, "layers.2.self_attn.k_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.2.self_attn.k_proj.trellis": { "shape": [ 320, 64, 64 ], "n_bytes": 2621440, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.2.self_attn.v_proj": { "stored_tensors": { "layers.2.self_attn.v_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.2.self_attn.v_proj.svh": { "shape": [ 1024 ], "n_bytes": 2048, "dtype": "torch.float16" }, "layers.2.self_attn.v_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.2.self_attn.v_proj.trellis": { "shape": [ 320, 64, 64 ], "n_bytes": 2621440, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.2.self_attn.o_proj": { "stored_tensors": { "layers.2.self_attn.o_proj.suh": { "shape": [ 4096 ], "n_bytes": 8192, "dtype": "torch.float16" }, "layers.2.self_attn.o_proj.svh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.2.self_attn.o_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.2.self_attn.o_proj.trellis": { "shape": [ 256, 320, 64 ], "n_bytes": 10485760, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.2.self_attn.q_norm": { "stored_tensors": { "layers.2.self_attn.q_norm.weight": { "shape": [ 128 ], "n_bytes": 256, "dtype": "torch.bfloat16" } } }, "layers.2.self_attn.k_norm": { "stored_tensors": { "layers.2.self_attn.k_norm.weight": { "shape": [ 128 ], "n_bytes": 256, "dtype": "torch.bfloat16" } } }, "layers.2.post_attention_layernorm": { "stored_tensors": { "layers.2.post_attention_layernorm.weight": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.bfloat16" } } }, "layers.2.mlp.up_proj": { "stored_tensors": { "layers.2.mlp.up_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.2.mlp.up_proj.svh": { "shape": [ 17408 ], "n_bytes": 34816, "dtype": "torch.float16" }, "layers.2.mlp.up_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.2.mlp.up_proj.trellis": { "shape": [ 320, 1088, 64 ], "n_bytes": 44564480, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.2.mlp.gate_proj": { "stored_tensors": { "layers.2.mlp.gate_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.2.mlp.gate_proj.svh": { "shape": [ 17408 ], "n_bytes": 34816, "dtype": "torch.float16" }, "layers.2.mlp.gate_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.2.mlp.gate_proj.trellis": { "shape": [ 320, 1088, 64 ], "n_bytes": 44564480, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.2.mlp.down_proj": { "stored_tensors": { "layers.2.mlp.down_proj.suh": { "shape": [ 17408 ], "n_bytes": 34816, "dtype": "torch.float16" }, "layers.2.mlp.down_proj.svh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.2.mlp.down_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.2.mlp.down_proj.trellis": { "shape": [ 1088, 320, 64 ], "n_bytes": 44564480, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.2.attention_conv.kernel_projection": { "stored_tensors": { "layers.2.attention_conv.kernel_projection.weight": { "shape": [ 1280, 5120 ], "n_bytes": 13107200, "dtype": "torch.float16" } } }, "layers.2.mlp_conv.kernel_projection": { "stored_tensors": { "layers.2.mlp_conv.kernel_projection.weight": { "shape": [ 1280, 5120 ], "n_bytes": 13107200, "dtype": "torch.float16" } } }, "layers.3.input_layernorm": { "stored_tensors": { "layers.3.input_layernorm.weight": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.bfloat16" } } }, "layers.3.self_attn.q_proj": { "stored_tensors": { "layers.3.self_attn.q_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.3.self_attn.q_proj.svh": { "shape": [ 4096 ], "n_bytes": 8192, "dtype": "torch.float16" }, "layers.3.self_attn.q_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.3.self_attn.q_proj.trellis": { "shape": [ 320, 256, 64 ], "n_bytes": 10485760, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.3.self_attn.k_proj": { "stored_tensors": { "layers.3.self_attn.k_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.3.self_attn.k_proj.svh": { "shape": [ 1024 ], "n_bytes": 2048, "dtype": "torch.float16" }, "layers.3.self_attn.k_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.3.self_attn.k_proj.trellis": { "shape": [ 320, 64, 64 ], "n_bytes": 2621440, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.3.self_attn.v_proj": { "stored_tensors": { "layers.3.self_attn.v_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.3.self_attn.v_proj.svh": { "shape": [ 1024 ], "n_bytes": 2048, "dtype": "torch.float16" }, "layers.3.self_attn.v_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.3.self_attn.v_proj.trellis": { "shape": [ 320, 64, 64 ], "n_bytes": 2621440, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.3.self_attn.o_proj": { "stored_tensors": { "layers.3.self_attn.o_proj.suh": { "shape": [ 4096 ], "n_bytes": 8192, "dtype": "torch.float16" }, "layers.3.self_attn.o_proj.svh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.3.self_attn.o_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.3.self_attn.o_proj.trellis": { "shape": [ 256, 320, 64 ], "n_bytes": 10485760, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.3.self_attn.q_norm": { "stored_tensors": { "layers.3.self_attn.q_norm.weight": { "shape": [ 128 ], "n_bytes": 256, "dtype": "torch.bfloat16" } } }, "layers.3.self_attn.k_norm": { "stored_tensors": { "layers.3.self_attn.k_norm.weight": { "shape": [ 128 ], "n_bytes": 256, "dtype": "torch.bfloat16" } } }, "layers.3.post_attention_layernorm": { "stored_tensors": { "layers.3.post_attention_layernorm.weight": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.bfloat16" } } }, "layers.3.mlp.up_proj": { "stored_tensors": { "layers.3.mlp.up_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.3.mlp.up_proj.svh": { "shape": [ 17408 ], "n_bytes": 34816, "dtype": "torch.float16" }, "layers.3.mlp.up_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.3.mlp.up_proj.trellis": { "shape": [ 320, 1088, 64 ], "n_bytes": 44564480, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.3.mlp.gate_proj": { "stored_tensors": { "layers.3.mlp.gate_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.3.mlp.gate_proj.svh": { "shape": [ 17408 ], "n_bytes": 34816, "dtype": "torch.float16" }, "layers.3.mlp.gate_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.3.mlp.gate_proj.trellis": { "shape": [ 320, 1088, 64 ], "n_bytes": 44564480, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.3.mlp.down_proj": { "stored_tensors": { "layers.3.mlp.down_proj.suh": { "shape": [ 17408 ], "n_bytes": 34816, "dtype": "torch.float16" }, "layers.3.mlp.down_proj.svh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.3.mlp.down_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.3.mlp.down_proj.trellis": { "shape": [ 1088, 320, 64 ], "n_bytes": 44564480, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.3.attention_conv.kernel_projection": { "stored_tensors": { "layers.3.attention_conv.kernel_projection.weight": { "shape": [ 1280, 5120 ], "n_bytes": 13107200, "dtype": "torch.float16" } } }, "layers.3.mlp_conv.kernel_projection": { "stored_tensors": { "layers.3.mlp_conv.kernel_projection.weight": { "shape": [ 1280, 5120 ], "n_bytes": 13107200, "dtype": "torch.float16" } } }, "layers.4.input_layernorm": { "stored_tensors": { "layers.4.input_layernorm.weight": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.bfloat16" } } }, "layers.4.self_attn.q_proj": { "stored_tensors": { "layers.4.self_attn.q_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.4.self_attn.q_proj.svh": { "shape": [ 4096 ], "n_bytes": 8192, "dtype": "torch.float16" }, "layers.4.self_attn.q_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.4.self_attn.q_proj.trellis": { "shape": [ 320, 256, 64 ], "n_bytes": 10485760, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.4.self_attn.k_proj": { "stored_tensors": { "layers.4.self_attn.k_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.4.self_attn.k_proj.svh": { "shape": [ 1024 ], "n_bytes": 2048, "dtype": "torch.float16" }, "layers.4.self_attn.k_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.4.self_attn.k_proj.trellis": { "shape": [ 320, 64, 64 ], "n_bytes": 2621440, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.4.self_attn.v_proj": { "stored_tensors": { "layers.4.self_attn.v_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.4.self_attn.v_proj.svh": { "shape": [ 1024 ], "n_bytes": 2048, "dtype": "torch.float16" }, "layers.4.self_attn.v_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.4.self_attn.v_proj.trellis": { "shape": [ 320, 64, 64 ], "n_bytes": 2621440, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.4.self_attn.o_proj": { "stored_tensors": { "layers.4.self_attn.o_proj.suh": { "shape": [ 4096 ], "n_bytes": 8192, "dtype": "torch.float16" }, "layers.4.self_attn.o_proj.svh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.4.self_attn.o_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.4.self_attn.o_proj.trellis": { "shape": [ 256, 320, 64 ], "n_bytes": 10485760, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.4.self_attn.q_norm": { "stored_tensors": { "layers.4.self_attn.q_norm.weight": { "shape": [ 128 ], "n_bytes": 256, "dtype": "torch.bfloat16" } } }, "layers.4.self_attn.k_norm": { "stored_tensors": { "layers.4.self_attn.k_norm.weight": { "shape": [ 128 ], "n_bytes": 256, "dtype": "torch.bfloat16" } } }, "layers.4.post_attention_layernorm": { "stored_tensors": { "layers.4.post_attention_layernorm.weight": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.bfloat16" } } }, "layers.4.mlp.up_proj": { "stored_tensors": { "layers.4.mlp.up_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.4.mlp.up_proj.svh": { "shape": [ 17408 ], "n_bytes": 34816, "dtype": "torch.float16" }, "layers.4.mlp.up_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.4.mlp.up_proj.trellis": { "shape": [ 320, 1088, 64 ], "n_bytes": 44564480, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.4.mlp.gate_proj": { "stored_tensors": { "layers.4.mlp.gate_proj.suh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.4.mlp.gate_proj.svh": { "shape": [ 17408 ], "n_bytes": 34816, "dtype": "torch.float16" }, "layers.4.mlp.gate_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.4.mlp.gate_proj.trellis": { "shape": [ 320, 1088, 64 ], "n_bytes": 44564480, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.4.mlp.down_proj": { "stored_tensors": { "layers.4.mlp.down_proj.suh": { "shape": [ 17408 ], "n_bytes": 34816, "dtype": "torch.float16" }, "layers.4.mlp.down_proj.svh": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.float16" }, "layers.4.mlp.down_proj.mul1": { "shape": [], "n_bytes": 4, "dtype": "torch.int32" }, "layers.4.mlp.down_proj.trellis": { "shape": [ 1088, 320, 64 ], "n_bytes": 44564480, "dtype": "torch.int16" } }, "quant_format": "exl3", "bits_per_weight": 4, "mul1_multiplier": 2212286765 }, "layers.4.attention_conv.kernel_projection": { "stored_tensors": { "layers.4.attention_conv.kernel_projection.weight": { "shape": [ 1280, 5120 ], "n_bytes": 13107200, "dtype": "torch.float16" } } }, "layers.4.mlp_conv.kernel_projection": { "stored_tensors": { "layers.4.mlp_conv.kernel_projection.weight": { "shape": [ 1280, 5120 ], "n_bytes": 13107200, "dtype": "torch.float16" } } }, "norm": { "stored_tensors": { "norm.weight": { "shape": [ 5120 ], "n_bytes": 10240, "dtype": "torch.bfloat16" } } }, "candidate_selector.hidden_projection": { "stored_tensors": { "candidate_selector.hidden_projection.weight": { "shape": [ 256, 5120 ], "n_bytes": 2621440, "dtype": "torch.float16" } } } } }