Llama-3.2-1B-Instruct-q4f16_1-MLC / ndarray-cache.json
CharlieFRuan's picture
Upload folder using huggingface_hub
7e6866e verified
{
"metadata": {
"ParamSize": 163,
"ParamBytes": 695242752.0,
"BitsPerParam": 4.500628909972241
},
"records": [
{
"dataPath": "params_shard_0.bin",
"format": "raw-shard",
"nbytes": 131334144,
"records": [
{
"name": "model.embed_tokens.q_weight",
"shape": [
128256,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 131334144,
"byteOffset": 0
}
],
"md5sum": "780f65bb3be7cdc499634d8cae98c5aa"
},
{
"dataPath": "params_shard_1.bin",
"format": "raw-shard",
"nbytes": 16777216,
"records": [
{
"name": "model.layers.0.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 0
}
],
"md5sum": "90cc2adb82e582599ddd8d97e5b1bcc0"
},
{
"dataPath": "params_shard_2.bin",
"format": "raw-shard",
"nbytes": 31498240,
"records": [
{
"name": "model.embed_tokens.q_scale",
"shape": [
128256,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 16416768,
"byteOffset": 0
},
{
"name": "model.layers.0.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 16416768
},
{
"name": "model.layers.0.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 16420864
},
{
"name": "model.layers.0.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 24809472
},
{
"name": "model.layers.0.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 25858048
},
{
"name": "model.layers.0.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 27955200
},
{
"name": "model.layers.0.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 27959296
},
{
"name": "model.layers.0.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 31105024
}
],
"md5sum": "d6f8e0150d3d538726eb1276cee29bd5"
},
{
"dataPath": "params_shard_3.bin",
"format": "raw-shard",
"nbytes": 30679040,
"records": [
{
"name": "model.layers.0.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.0.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 2097152
},
{
"name": "model.layers.1.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2359296
},
{
"name": "model.layers.1.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 2363392
},
{
"name": "model.layers.1.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 10752000
},
{
"name": "model.layers.1.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 11800576
},
{
"name": "model.layers.1.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 28577792
},
{
"name": "model.layers.1.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 30674944
}
],
"md5sum": "2760a8b4dcb645e26283def0403f4deb"
},
{
"dataPath": "params_shard_4.bin",
"format": "raw-shard",
"nbytes": 32116736,
"records": [
{
"name": "model.layers.1.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "model.layers.1.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 3145728
},
{
"name": "model.layers.1.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 3538944
},
{
"name": "model.layers.1.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 5636096
},
{
"name": "model.layers.10.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 5898240
},
{
"name": "model.layers.10.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 5902336
},
{
"name": "model.layers.10.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 14290944
},
{
"name": "model.layers.10.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 15339520
}
],
"md5sum": "d40c880de01576bbaecc3abf7f820688"
},
{
"dataPath": "params_shard_5.bin",
"format": "raw-shard",
"nbytes": 16777216,
"records": [
{
"name": "model.layers.11.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 0
}
],
"md5sum": "9e3b2f06d748bb4267a77b44d8a489df"
},
{
"dataPath": "params_shard_6.bin",
"format": "raw-shard",
"nbytes": 25444352,
"records": [
{
"name": "model.layers.10.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.10.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2097152
},
{
"name": "model.layers.10.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 2101248
},
{
"name": "model.layers.10.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 5246976
},
{
"name": "model.layers.10.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 5640192
},
{
"name": "model.layers.10.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 7737344
},
{
"name": "model.layers.11.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 7999488
},
{
"name": "model.layers.11.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 8003584
},
{
"name": "model.layers.11.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 16392192
},
{
"name": "model.layers.11.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 17440768
},
{
"name": "model.layers.11.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 19537920
},
{
"name": "model.layers.11.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 19542016
},
{
"name": "model.layers.11.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 22687744
},
{
"name": "model.layers.11.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 23080960
},
{
"name": "model.layers.11.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 25178112
},
{
"name": "model.layers.12.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 25440256
}
],
"md5sum": "cba2e299b4275ebc7054f8ad16b7cdb9"
},
{
"dataPath": "params_shard_7.bin",
"format": "raw-shard",
"nbytes": 31854592,
"records": [
{
"name": "model.layers.12.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.12.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.12.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 9437184
},
{
"name": "model.layers.12.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 26214400
},
{
"name": "model.layers.12.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 28311552
},
{
"name": "model.layers.12.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 28315648
},
{
"name": "model.layers.12.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 31461376
}
],
"md5sum": "2bc3abb018d2d9db8a546c24c33abbc2"
},
{
"dataPath": "params_shard_8.bin",
"format": "raw-shard",
"nbytes": 30679040,
"records": [
{
"name": "model.layers.12.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.12.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 2097152
},
{
"name": "model.layers.13.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2359296
},
{
"name": "model.layers.13.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 2363392
},
{
"name": "model.layers.13.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 10752000
},
{
"name": "model.layers.13.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 11800576
},
{
"name": "model.layers.13.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 28577792
},
{
"name": "model.layers.13.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 30674944
}
],
"md5sum": "4f50bccecdd051f00c706d50db12cda6"
},
{
"dataPath": "params_shard_9.bin",
"format": "raw-shard",
"nbytes": 32116736,
"records": [
{
"name": "model.layers.13.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "model.layers.13.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 3145728
},
{
"name": "model.layers.13.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 3538944
},
{
"name": "model.layers.13.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 5636096
},
{
"name": "model.layers.14.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 5898240
},
{
"name": "model.layers.14.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 5902336
},
{
"name": "model.layers.14.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 14290944
},
{
"name": "model.layers.14.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 15339520
}
],
"md5sum": "2b7c430520a6507126e0bbb9b31bb7ee"
},
{
"dataPath": "params_shard_10.bin",
"format": "raw-shard",
"nbytes": 16777216,
"records": [
{
"name": "model.layers.15.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 0
}
],
"md5sum": "043ab15ecb84e64c808edbfe036ae595"
},
{
"dataPath": "params_shard_11.bin",
"format": "raw-shard",
"nbytes": 25444352,
"records": [
{
"name": "model.layers.14.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.14.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2097152
},
{
"name": "model.layers.14.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 2101248
},
{
"name": "model.layers.14.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 5246976
},
{
"name": "model.layers.14.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 5640192
},
{
"name": "model.layers.14.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 7737344
},
{
"name": "model.layers.15.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 7999488
},
{
"name": "model.layers.15.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 8003584
},
{
"name": "model.layers.15.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 16392192
},
{
"name": "model.layers.15.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 17440768
},
{
"name": "model.layers.15.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 19537920
},
{
"name": "model.layers.15.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 19542016
},
{
"name": "model.layers.15.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 22687744
},
{
"name": "model.layers.15.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 23080960
},
{
"name": "model.layers.15.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 25178112
},
{
"name": "model.layers.2.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 25440256
}
],
"md5sum": "542c8599d640c38ae6f192a742e235d0"
},
{
"dataPath": "params_shard_12.bin",
"format": "raw-shard",
"nbytes": 31854592,
"records": [
{
"name": "model.layers.2.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.2.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.2.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 9437184
},
{
"name": "model.layers.2.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 26214400
},
{
"name": "model.layers.2.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 28311552
},
{
"name": "model.layers.2.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 28315648
},
{
"name": "model.layers.2.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 31461376
}
],
"md5sum": "b4cd5568e2877aabade1a08c4f9f8e20"
},
{
"dataPath": "params_shard_13.bin",
"format": "raw-shard",
"nbytes": 30679040,
"records": [
{
"name": "model.layers.2.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.2.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 2097152
},
{
"name": "model.layers.3.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2359296
},
{
"name": "model.layers.3.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 2363392
},
{
"name": "model.layers.3.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 10752000
},
{
"name": "model.layers.3.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 11800576
},
{
"name": "model.layers.3.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 28577792
},
{
"name": "model.layers.3.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 30674944
}
],
"md5sum": "cbce2e42eb5932d41213ab17b83daed4"
},
{
"dataPath": "params_shard_14.bin",
"format": "raw-shard",
"nbytes": 32116736,
"records": [
{
"name": "model.layers.3.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "model.layers.3.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 3145728
},
{
"name": "model.layers.3.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 3538944
},
{
"name": "model.layers.3.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 5636096
},
{
"name": "model.layers.4.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 5898240
},
{
"name": "model.layers.4.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 5902336
},
{
"name": "model.layers.4.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 14290944
},
{
"name": "model.layers.4.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 15339520
}
],
"md5sum": "d7f189698bd8fc3341223af3b608f4d5"
},
{
"dataPath": "params_shard_15.bin",
"format": "raw-shard",
"nbytes": 16777216,
"records": [
{
"name": "model.layers.5.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 0
}
],
"md5sum": "77dc740bbecc7ab456e27ba134d17034"
},
{
"dataPath": "params_shard_16.bin",
"format": "raw-shard",
"nbytes": 25444352,
"records": [
{
"name": "model.layers.4.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.4.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2097152
},
{
"name": "model.layers.4.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 2101248
},
{
"name": "model.layers.4.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 5246976
},
{
"name": "model.layers.4.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 5640192
},
{
"name": "model.layers.4.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 7737344
},
{
"name": "model.layers.5.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 7999488
},
{
"name": "model.layers.5.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 8003584
},
{
"name": "model.layers.5.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 16392192
},
{
"name": "model.layers.5.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 17440768
},
{
"name": "model.layers.5.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 19537920
},
{
"name": "model.layers.5.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 19542016
},
{
"name": "model.layers.5.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 22687744
},
{
"name": "model.layers.5.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 23080960
},
{
"name": "model.layers.5.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 25178112
},
{
"name": "model.layers.6.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 25440256
}
],
"md5sum": "03cb14edc29900dbc711ebb144ad3705"
},
{
"dataPath": "params_shard_17.bin",
"format": "raw-shard",
"nbytes": 31854592,
"records": [
{
"name": "model.layers.6.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 0
},
{
"name": "model.layers.6.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 8388608
},
{
"name": "model.layers.6.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 9437184
},
{
"name": "model.layers.6.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 26214400
},
{
"name": "model.layers.6.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 28311552
},
{
"name": "model.layers.6.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 28315648
},
{
"name": "model.layers.6.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 31461376
}
],
"md5sum": "e8ac271894487d2d5d0785ccace377cd"
},
{
"dataPath": "params_shard_18.bin",
"format": "raw-shard",
"nbytes": 30679040,
"records": [
{
"name": "model.layers.6.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.6.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 2097152
},
{
"name": "model.layers.7.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2359296
},
{
"name": "model.layers.7.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 2363392
},
{
"name": "model.layers.7.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 10752000
},
{
"name": "model.layers.7.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 11800576
},
{
"name": "model.layers.7.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 28577792
},
{
"name": "model.layers.7.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 30674944
}
],
"md5sum": "ba970b7d1de3b1f9319ab0083b0f5dbe"
},
{
"dataPath": "params_shard_19.bin",
"format": "raw-shard",
"nbytes": 32116736,
"records": [
{
"name": "model.layers.7.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 0
},
{
"name": "model.layers.7.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 3145728
},
{
"name": "model.layers.7.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 3538944
},
{
"name": "model.layers.7.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 5636096
},
{
"name": "model.layers.8.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 5898240
},
{
"name": "model.layers.8.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 5902336
},
{
"name": "model.layers.8.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 14290944
},
{
"name": "model.layers.8.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 15339520
}
],
"md5sum": "a2a7115d600451c223e73542f33715c2"
},
{
"dataPath": "params_shard_20.bin",
"format": "raw-shard",
"nbytes": 16777216,
"records": [
{
"name": "model.layers.9.mlp.gate_up_proj.q_weight",
"shape": [
16384,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 16777216,
"byteOffset": 0
}
],
"md5sum": "e4a1b907f3c17eb89b314cd1d09f27cd"
},
{
"dataPath": "params_shard_21.bin",
"format": "raw-shard",
"nbytes": 25444352,
"records": [
{
"name": "model.layers.8.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 0
},
{
"name": "model.layers.8.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 2097152
},
{
"name": "model.layers.8.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 2101248
},
{
"name": "model.layers.8.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 5246976
},
{
"name": "model.layers.8.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 5640192
},
{
"name": "model.layers.8.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 7737344
},
{
"name": "model.layers.9.input_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 7999488
},
{
"name": "model.layers.9.mlp.down_proj.q_weight",
"shape": [
2048,
1024
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 8388608,
"byteOffset": 8003584
},
{
"name": "model.layers.9.mlp.down_proj.q_scale",
"shape": [
2048,
256
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 1048576,
"byteOffset": 16392192
},
{
"name": "model.layers.9.mlp.gate_up_proj.q_scale",
"shape": [
16384,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 17440768
},
{
"name": "model.layers.9.post_attention_layernorm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 19537920
},
{
"name": "model.layers.9.self_attn.qkv_proj.q_weight",
"shape": [
3072,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3145728,
"byteOffset": 19542016
},
{
"name": "model.layers.9.self_attn.qkv_proj.q_scale",
"shape": [
3072,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 393216,
"byteOffset": 22687744
},
{
"name": "model.layers.9.self_attn.o_proj.q_weight",
"shape": [
2048,
256
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 2097152,
"byteOffset": 23080960
},
{
"name": "model.layers.9.self_attn.o_proj.q_scale",
"shape": [
2048,
64
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 262144,
"byteOffset": 25178112
},
{
"name": "model.norm.weight",
"shape": [
2048
],
"dtype": "float16",
"format": "f32-to-bf16",
"nbytes": 4096,
"byteOffset": 25440256
}
],
"md5sum": "cb6d8c6f94323219bafcc299b20bf8a8"
}
]
}