| { |
| "metadata": { |
| "ParamSize": 163, |
| "ParamBytes": 695245056.0, |
| "BitsPerParam": 4.500628907887781 |
| }, |
| "records": [ |
| { |
| "dataPath": "params_shard_0.bin", |
| "format": "raw-shard", |
| "nbytes": 131336192, |
| "records": [ |
| { |
| "name": "model.embed_tokens.q_weight", |
| "shape": [ |
| 128258, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 131336192, |
| "byteOffset": 0 |
| } |
| ], |
| "md5sum": "6124bad0e973c9ca2eb02850856705b9" |
| }, |
| { |
| "dataPath": "params_shard_1.bin", |
| "format": "raw-shard", |
| "nbytes": 16777216, |
| "records": [ |
| { |
| "name": "model.layers.0.mlp.gate_up_proj.q_weight", |
| "shape": [ |
| 16384, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 16777216, |
| "byteOffset": 0 |
| } |
| ], |
| "md5sum": "73428a8effc04d761eed7840fdf6b8c7" |
| }, |
| { |
| "dataPath": "params_shard_2.bin", |
| "format": "raw-shard", |
| "nbytes": 31498496, |
| "records": [ |
| { |
| "name": "model.embed_tokens.q_scale", |
| "shape": [ |
| 128258, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 16417024, |
| "byteOffset": 0 |
| }, |
| { |
| "name": "model.layers.0.input_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 16417024 |
| }, |
| { |
| "name": "model.layers.0.mlp.down_proj.q_weight", |
| "shape": [ |
| 2048, |
| 1024 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 8388608, |
| "byteOffset": 16421120 |
| }, |
| { |
| "name": "model.layers.0.mlp.down_proj.q_scale", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 1048576, |
| "byteOffset": 24809728 |
| }, |
| { |
| "name": "model.layers.0.mlp.gate_up_proj.q_scale", |
| "shape": [ |
| 16384, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 25858304 |
| }, |
| { |
| "name": "model.layers.0.post_attention_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 27955456 |
| }, |
| { |
| "name": "model.layers.0.self_attn.qkv_proj.q_weight", |
| "shape": [ |
| 3072, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 3145728, |
| "byteOffset": 27959552 |
| }, |
| { |
| "name": "model.layers.0.self_attn.qkv_proj.q_scale", |
| "shape": [ |
| 3072, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 393216, |
| "byteOffset": 31105280 |
| } |
| ], |
| "md5sum": "a857d6062df7e6034a3f00428a6eba4a" |
| }, |
| { |
| "dataPath": "params_shard_3.bin", |
| "format": "raw-shard", |
| "nbytes": 30679040, |
| "records": [ |
| { |
| "name": "model.layers.0.self_attn.o_proj.q_weight", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 0 |
| }, |
| { |
| "name": "model.layers.0.self_attn.o_proj.q_scale", |
| "shape": [ |
| 2048, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 262144, |
| "byteOffset": 2097152 |
| }, |
| { |
| "name": "model.layers.1.input_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 2359296 |
| }, |
| { |
| "name": "model.layers.1.mlp.down_proj.q_weight", |
| "shape": [ |
| 2048, |
| 1024 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 8388608, |
| "byteOffset": 2363392 |
| }, |
| { |
| "name": "model.layers.1.mlp.down_proj.q_scale", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 1048576, |
| "byteOffset": 10752000 |
| }, |
| { |
| "name": "model.layers.1.mlp.gate_up_proj.q_weight", |
| "shape": [ |
| 16384, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 16777216, |
| "byteOffset": 11800576 |
| }, |
| { |
| "name": "model.layers.1.mlp.gate_up_proj.q_scale", |
| "shape": [ |
| 16384, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 28577792 |
| }, |
| { |
| "name": "model.layers.1.post_attention_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 30674944 |
| } |
| ], |
| "md5sum": "35880f151a0320540423f6a56b0e39b0" |
| }, |
| { |
| "dataPath": "params_shard_4.bin", |
| "format": "raw-shard", |
| "nbytes": 32116736, |
| "records": [ |
| { |
| "name": "model.layers.1.self_attn.qkv_proj.q_weight", |
| "shape": [ |
| 3072, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 3145728, |
| "byteOffset": 0 |
| }, |
| { |
| "name": "model.layers.1.self_attn.qkv_proj.q_scale", |
| "shape": [ |
| 3072, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 393216, |
| "byteOffset": 3145728 |
| }, |
| { |
| "name": "model.layers.1.self_attn.o_proj.q_weight", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 3538944 |
| }, |
| { |
| "name": "model.layers.1.self_attn.o_proj.q_scale", |
| "shape": [ |
| 2048, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 262144, |
| "byteOffset": 5636096 |
| }, |
| { |
| "name": "model.layers.10.input_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 5898240 |
| }, |
| { |
| "name": "model.layers.10.mlp.down_proj.q_weight", |
| "shape": [ |
| 2048, |
| 1024 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 8388608, |
| "byteOffset": 5902336 |
| }, |
| { |
| "name": "model.layers.10.mlp.down_proj.q_scale", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 1048576, |
| "byteOffset": 14290944 |
| }, |
| { |
| "name": "model.layers.10.mlp.gate_up_proj.q_weight", |
| "shape": [ |
| 16384, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 16777216, |
| "byteOffset": 15339520 |
| } |
| ], |
| "md5sum": "0399e7018251ef272ce9e454be12e330" |
| }, |
| { |
| "dataPath": "params_shard_5.bin", |
| "format": "raw-shard", |
| "nbytes": 16777216, |
| "records": [ |
| { |
| "name": "model.layers.11.mlp.gate_up_proj.q_weight", |
| "shape": [ |
| 16384, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 16777216, |
| "byteOffset": 0 |
| } |
| ], |
| "md5sum": "d16e6c34f09f4a237b34a03bb76865df" |
| }, |
| { |
| "dataPath": "params_shard_6.bin", |
| "format": "raw-shard", |
| "nbytes": 25444352, |
| "records": [ |
| { |
| "name": "model.layers.10.mlp.gate_up_proj.q_scale", |
| "shape": [ |
| 16384, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 0 |
| }, |
| { |
| "name": "model.layers.10.post_attention_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 2097152 |
| }, |
| { |
| "name": "model.layers.10.self_attn.qkv_proj.q_weight", |
| "shape": [ |
| 3072, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 3145728, |
| "byteOffset": 2101248 |
| }, |
| { |
| "name": "model.layers.10.self_attn.qkv_proj.q_scale", |
| "shape": [ |
| 3072, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 393216, |
| "byteOffset": 5246976 |
| }, |
| { |
| "name": "model.layers.10.self_attn.o_proj.q_weight", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 5640192 |
| }, |
| { |
| "name": "model.layers.10.self_attn.o_proj.q_scale", |
| "shape": [ |
| 2048, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 262144, |
| "byteOffset": 7737344 |
| }, |
| { |
| "name": "model.layers.11.input_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 7999488 |
| }, |
| { |
| "name": "model.layers.11.mlp.down_proj.q_weight", |
| "shape": [ |
| 2048, |
| 1024 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 8388608, |
| "byteOffset": 8003584 |
| }, |
| { |
| "name": "model.layers.11.mlp.down_proj.q_scale", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 1048576, |
| "byteOffset": 16392192 |
| }, |
| { |
| "name": "model.layers.11.mlp.gate_up_proj.q_scale", |
| "shape": [ |
| 16384, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 17440768 |
| }, |
| { |
| "name": "model.layers.11.post_attention_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 19537920 |
| }, |
| { |
| "name": "model.layers.11.self_attn.qkv_proj.q_weight", |
| "shape": [ |
| 3072, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 3145728, |
| "byteOffset": 19542016 |
| }, |
| { |
| "name": "model.layers.11.self_attn.qkv_proj.q_scale", |
| "shape": [ |
| 3072, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 393216, |
| "byteOffset": 22687744 |
| }, |
| { |
| "name": "model.layers.11.self_attn.o_proj.q_weight", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 23080960 |
| }, |
| { |
| "name": "model.layers.11.self_attn.o_proj.q_scale", |
| "shape": [ |
| 2048, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 262144, |
| "byteOffset": 25178112 |
| }, |
| { |
| "name": "model.layers.12.input_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 25440256 |
| } |
| ], |
| "md5sum": "89a1adc7f1d1810c4676d17d0438d17c" |
| }, |
| { |
| "dataPath": "params_shard_7.bin", |
| "format": "raw-shard", |
| "nbytes": 31854592, |
| "records": [ |
| { |
| "name": "model.layers.12.mlp.down_proj.q_weight", |
| "shape": [ |
| 2048, |
| 1024 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 8388608, |
| "byteOffset": 0 |
| }, |
| { |
| "name": "model.layers.12.mlp.down_proj.q_scale", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 1048576, |
| "byteOffset": 8388608 |
| }, |
| { |
| "name": "model.layers.12.mlp.gate_up_proj.q_weight", |
| "shape": [ |
| 16384, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 16777216, |
| "byteOffset": 9437184 |
| }, |
| { |
| "name": "model.layers.12.mlp.gate_up_proj.q_scale", |
| "shape": [ |
| 16384, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 26214400 |
| }, |
| { |
| "name": "model.layers.12.post_attention_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 28311552 |
| }, |
| { |
| "name": "model.layers.12.self_attn.qkv_proj.q_weight", |
| "shape": [ |
| 3072, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 3145728, |
| "byteOffset": 28315648 |
| }, |
| { |
| "name": "model.layers.12.self_attn.qkv_proj.q_scale", |
| "shape": [ |
| 3072, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 393216, |
| "byteOffset": 31461376 |
| } |
| ], |
| "md5sum": "580e966efc6dccccff02143a3ac6ea93" |
| }, |
| { |
| "dataPath": "params_shard_8.bin", |
| "format": "raw-shard", |
| "nbytes": 30679040, |
| "records": [ |
| { |
| "name": "model.layers.12.self_attn.o_proj.q_weight", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 0 |
| }, |
| { |
| "name": "model.layers.12.self_attn.o_proj.q_scale", |
| "shape": [ |
| 2048, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 262144, |
| "byteOffset": 2097152 |
| }, |
| { |
| "name": "model.layers.13.input_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 2359296 |
| }, |
| { |
| "name": "model.layers.13.mlp.down_proj.q_weight", |
| "shape": [ |
| 2048, |
| 1024 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 8388608, |
| "byteOffset": 2363392 |
| }, |
| { |
| "name": "model.layers.13.mlp.down_proj.q_scale", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 1048576, |
| "byteOffset": 10752000 |
| }, |
| { |
| "name": "model.layers.13.mlp.gate_up_proj.q_weight", |
| "shape": [ |
| 16384, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 16777216, |
| "byteOffset": 11800576 |
| }, |
| { |
| "name": "model.layers.13.mlp.gate_up_proj.q_scale", |
| "shape": [ |
| 16384, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 28577792 |
| }, |
| { |
| "name": "model.layers.13.post_attention_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 30674944 |
| } |
| ], |
| "md5sum": "5d0ff86eb3582a01fc5cc8ff05994bac" |
| }, |
| { |
| "dataPath": "params_shard_9.bin", |
| "format": "raw-shard", |
| "nbytes": 32116736, |
| "records": [ |
| { |
| "name": "model.layers.13.self_attn.qkv_proj.q_weight", |
| "shape": [ |
| 3072, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 3145728, |
| "byteOffset": 0 |
| }, |
| { |
| "name": "model.layers.13.self_attn.qkv_proj.q_scale", |
| "shape": [ |
| 3072, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 393216, |
| "byteOffset": 3145728 |
| }, |
| { |
| "name": "model.layers.13.self_attn.o_proj.q_weight", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 3538944 |
| }, |
| { |
| "name": "model.layers.13.self_attn.o_proj.q_scale", |
| "shape": [ |
| 2048, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 262144, |
| "byteOffset": 5636096 |
| }, |
| { |
| "name": "model.layers.14.input_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 5898240 |
| }, |
| { |
| "name": "model.layers.14.mlp.down_proj.q_weight", |
| "shape": [ |
| 2048, |
| 1024 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 8388608, |
| "byteOffset": 5902336 |
| }, |
| { |
| "name": "model.layers.14.mlp.down_proj.q_scale", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 1048576, |
| "byteOffset": 14290944 |
| }, |
| { |
| "name": "model.layers.14.mlp.gate_up_proj.q_weight", |
| "shape": [ |
| 16384, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 16777216, |
| "byteOffset": 15339520 |
| } |
| ], |
| "md5sum": "1069869e7d8fc765ba31c9b2649c7e97" |
| }, |
| { |
| "dataPath": "params_shard_10.bin", |
| "format": "raw-shard", |
| "nbytes": 16777216, |
| "records": [ |
| { |
| "name": "model.layers.15.mlp.gate_up_proj.q_weight", |
| "shape": [ |
| 16384, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 16777216, |
| "byteOffset": 0 |
| } |
| ], |
| "md5sum": "40bdf3661bdbdf6287d8e6b384a8106b" |
| }, |
| { |
| "dataPath": "params_shard_11.bin", |
| "format": "raw-shard", |
| "nbytes": 25444352, |
| "records": [ |
| { |
| "name": "model.layers.14.mlp.gate_up_proj.q_scale", |
| "shape": [ |
| 16384, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 0 |
| }, |
| { |
| "name": "model.layers.14.post_attention_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 2097152 |
| }, |
| { |
| "name": "model.layers.14.self_attn.qkv_proj.q_weight", |
| "shape": [ |
| 3072, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 3145728, |
| "byteOffset": 2101248 |
| }, |
| { |
| "name": "model.layers.14.self_attn.qkv_proj.q_scale", |
| "shape": [ |
| 3072, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 393216, |
| "byteOffset": 5246976 |
| }, |
| { |
| "name": "model.layers.14.self_attn.o_proj.q_weight", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 5640192 |
| }, |
| { |
| "name": "model.layers.14.self_attn.o_proj.q_scale", |
| "shape": [ |
| 2048, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 262144, |
| "byteOffset": 7737344 |
| }, |
| { |
| "name": "model.layers.15.input_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 7999488 |
| }, |
| { |
| "name": "model.layers.15.mlp.down_proj.q_weight", |
| "shape": [ |
| 2048, |
| 1024 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 8388608, |
| "byteOffset": 8003584 |
| }, |
| { |
| "name": "model.layers.15.mlp.down_proj.q_scale", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 1048576, |
| "byteOffset": 16392192 |
| }, |
| { |
| "name": "model.layers.15.mlp.gate_up_proj.q_scale", |
| "shape": [ |
| 16384, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 17440768 |
| }, |
| { |
| "name": "model.layers.15.post_attention_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 19537920 |
| }, |
| { |
| "name": "model.layers.15.self_attn.qkv_proj.q_weight", |
| "shape": [ |
| 3072, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 3145728, |
| "byteOffset": 19542016 |
| }, |
| { |
| "name": "model.layers.15.self_attn.qkv_proj.q_scale", |
| "shape": [ |
| 3072, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 393216, |
| "byteOffset": 22687744 |
| }, |
| { |
| "name": "model.layers.15.self_attn.o_proj.q_weight", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 23080960 |
| }, |
| { |
| "name": "model.layers.15.self_attn.o_proj.q_scale", |
| "shape": [ |
| 2048, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 262144, |
| "byteOffset": 25178112 |
| }, |
| { |
| "name": "model.layers.2.input_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 25440256 |
| } |
| ], |
| "md5sum": "99b309c374b5fcd2fa939c6823f16719" |
| }, |
| { |
| "dataPath": "params_shard_12.bin", |
| "format": "raw-shard", |
| "nbytes": 31854592, |
| "records": [ |
| { |
| "name": "model.layers.2.mlp.down_proj.q_weight", |
| "shape": [ |
| 2048, |
| 1024 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 8388608, |
| "byteOffset": 0 |
| }, |
| { |
| "name": "model.layers.2.mlp.down_proj.q_scale", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 1048576, |
| "byteOffset": 8388608 |
| }, |
| { |
| "name": "model.layers.2.mlp.gate_up_proj.q_weight", |
| "shape": [ |
| 16384, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 16777216, |
| "byteOffset": 9437184 |
| }, |
| { |
| "name": "model.layers.2.mlp.gate_up_proj.q_scale", |
| "shape": [ |
| 16384, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 26214400 |
| }, |
| { |
| "name": "model.layers.2.post_attention_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 28311552 |
| }, |
| { |
| "name": "model.layers.2.self_attn.qkv_proj.q_weight", |
| "shape": [ |
| 3072, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 3145728, |
| "byteOffset": 28315648 |
| }, |
| { |
| "name": "model.layers.2.self_attn.qkv_proj.q_scale", |
| "shape": [ |
| 3072, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 393216, |
| "byteOffset": 31461376 |
| } |
| ], |
| "md5sum": "cb9bd7c0736545c9df8195c45184af09" |
| }, |
| { |
| "dataPath": "params_shard_13.bin", |
| "format": "raw-shard", |
| "nbytes": 30679040, |
| "records": [ |
| { |
| "name": "model.layers.2.self_attn.o_proj.q_weight", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 0 |
| }, |
| { |
| "name": "model.layers.2.self_attn.o_proj.q_scale", |
| "shape": [ |
| 2048, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 262144, |
| "byteOffset": 2097152 |
| }, |
| { |
| "name": "model.layers.3.input_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 2359296 |
| }, |
| { |
| "name": "model.layers.3.mlp.down_proj.q_weight", |
| "shape": [ |
| 2048, |
| 1024 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 8388608, |
| "byteOffset": 2363392 |
| }, |
| { |
| "name": "model.layers.3.mlp.down_proj.q_scale", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 1048576, |
| "byteOffset": 10752000 |
| }, |
| { |
| "name": "model.layers.3.mlp.gate_up_proj.q_weight", |
| "shape": [ |
| 16384, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 16777216, |
| "byteOffset": 11800576 |
| }, |
| { |
| "name": "model.layers.3.mlp.gate_up_proj.q_scale", |
| "shape": [ |
| 16384, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 28577792 |
| }, |
| { |
| "name": "model.layers.3.post_attention_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 30674944 |
| } |
| ], |
| "md5sum": "ad094798695790f88b895530893f4026" |
| }, |
| { |
| "dataPath": "params_shard_14.bin", |
| "format": "raw-shard", |
| "nbytes": 32116736, |
| "records": [ |
| { |
| "name": "model.layers.3.self_attn.qkv_proj.q_weight", |
| "shape": [ |
| 3072, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 3145728, |
| "byteOffset": 0 |
| }, |
| { |
| "name": "model.layers.3.self_attn.qkv_proj.q_scale", |
| "shape": [ |
| 3072, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 393216, |
| "byteOffset": 3145728 |
| }, |
| { |
| "name": "model.layers.3.self_attn.o_proj.q_weight", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 3538944 |
| }, |
| { |
| "name": "model.layers.3.self_attn.o_proj.q_scale", |
| "shape": [ |
| 2048, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 262144, |
| "byteOffset": 5636096 |
| }, |
| { |
| "name": "model.layers.4.input_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 5898240 |
| }, |
| { |
| "name": "model.layers.4.mlp.down_proj.q_weight", |
| "shape": [ |
| 2048, |
| 1024 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 8388608, |
| "byteOffset": 5902336 |
| }, |
| { |
| "name": "model.layers.4.mlp.down_proj.q_scale", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 1048576, |
| "byteOffset": 14290944 |
| }, |
| { |
| "name": "model.layers.4.mlp.gate_up_proj.q_weight", |
| "shape": [ |
| 16384, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 16777216, |
| "byteOffset": 15339520 |
| } |
| ], |
| "md5sum": "607563464d8a59cae7c60da0950c56a0" |
| }, |
| { |
| "dataPath": "params_shard_15.bin", |
| "format": "raw-shard", |
| "nbytes": 16777216, |
| "records": [ |
| { |
| "name": "model.layers.5.mlp.gate_up_proj.q_weight", |
| "shape": [ |
| 16384, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 16777216, |
| "byteOffset": 0 |
| } |
| ], |
| "md5sum": "b86b212a485e75ab5577f7fce6f0e891" |
| }, |
| { |
| "dataPath": "params_shard_16.bin", |
| "format": "raw-shard", |
| "nbytes": 25444352, |
| "records": [ |
| { |
| "name": "model.layers.4.mlp.gate_up_proj.q_scale", |
| "shape": [ |
| 16384, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 0 |
| }, |
| { |
| "name": "model.layers.4.post_attention_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 2097152 |
| }, |
| { |
| "name": "model.layers.4.self_attn.qkv_proj.q_weight", |
| "shape": [ |
| 3072, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 3145728, |
| "byteOffset": 2101248 |
| }, |
| { |
| "name": "model.layers.4.self_attn.qkv_proj.q_scale", |
| "shape": [ |
| 3072, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 393216, |
| "byteOffset": 5246976 |
| }, |
| { |
| "name": "model.layers.4.self_attn.o_proj.q_weight", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 5640192 |
| }, |
| { |
| "name": "model.layers.4.self_attn.o_proj.q_scale", |
| "shape": [ |
| 2048, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 262144, |
| "byteOffset": 7737344 |
| }, |
| { |
| "name": "model.layers.5.input_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 7999488 |
| }, |
| { |
| "name": "model.layers.5.mlp.down_proj.q_weight", |
| "shape": [ |
| 2048, |
| 1024 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 8388608, |
| "byteOffset": 8003584 |
| }, |
| { |
| "name": "model.layers.5.mlp.down_proj.q_scale", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 1048576, |
| "byteOffset": 16392192 |
| }, |
| { |
| "name": "model.layers.5.mlp.gate_up_proj.q_scale", |
| "shape": [ |
| 16384, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 17440768 |
| }, |
| { |
| "name": "model.layers.5.post_attention_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 19537920 |
| }, |
| { |
| "name": "model.layers.5.self_attn.qkv_proj.q_weight", |
| "shape": [ |
| 3072, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 3145728, |
| "byteOffset": 19542016 |
| }, |
| { |
| "name": "model.layers.5.self_attn.qkv_proj.q_scale", |
| "shape": [ |
| 3072, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 393216, |
| "byteOffset": 22687744 |
| }, |
| { |
| "name": "model.layers.5.self_attn.o_proj.q_weight", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 23080960 |
| }, |
| { |
| "name": "model.layers.5.self_attn.o_proj.q_scale", |
| "shape": [ |
| 2048, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 262144, |
| "byteOffset": 25178112 |
| }, |
| { |
| "name": "model.layers.6.input_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 25440256 |
| } |
| ], |
| "md5sum": "c88417f337ecbaeb8a7ac02535e9c5a1" |
| }, |
| { |
| "dataPath": "params_shard_17.bin", |
| "format": "raw-shard", |
| "nbytes": 31854592, |
| "records": [ |
| { |
| "name": "model.layers.6.mlp.down_proj.q_weight", |
| "shape": [ |
| 2048, |
| 1024 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 8388608, |
| "byteOffset": 0 |
| }, |
| { |
| "name": "model.layers.6.mlp.down_proj.q_scale", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 1048576, |
| "byteOffset": 8388608 |
| }, |
| { |
| "name": "model.layers.6.mlp.gate_up_proj.q_weight", |
| "shape": [ |
| 16384, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 16777216, |
| "byteOffset": 9437184 |
| }, |
| { |
| "name": "model.layers.6.mlp.gate_up_proj.q_scale", |
| "shape": [ |
| 16384, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 26214400 |
| }, |
| { |
| "name": "model.layers.6.post_attention_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 28311552 |
| }, |
| { |
| "name": "model.layers.6.self_attn.qkv_proj.q_weight", |
| "shape": [ |
| 3072, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 3145728, |
| "byteOffset": 28315648 |
| }, |
| { |
| "name": "model.layers.6.self_attn.qkv_proj.q_scale", |
| "shape": [ |
| 3072, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 393216, |
| "byteOffset": 31461376 |
| } |
| ], |
| "md5sum": "4ceff3c4fa259163a596f446a34bfa90" |
| }, |
| { |
| "dataPath": "params_shard_18.bin", |
| "format": "raw-shard", |
| "nbytes": 30679040, |
| "records": [ |
| { |
| "name": "model.layers.6.self_attn.o_proj.q_weight", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 0 |
| }, |
| { |
| "name": "model.layers.6.self_attn.o_proj.q_scale", |
| "shape": [ |
| 2048, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 262144, |
| "byteOffset": 2097152 |
| }, |
| { |
| "name": "model.layers.7.input_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 2359296 |
| }, |
| { |
| "name": "model.layers.7.mlp.down_proj.q_weight", |
| "shape": [ |
| 2048, |
| 1024 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 8388608, |
| "byteOffset": 2363392 |
| }, |
| { |
| "name": "model.layers.7.mlp.down_proj.q_scale", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 1048576, |
| "byteOffset": 10752000 |
| }, |
| { |
| "name": "model.layers.7.mlp.gate_up_proj.q_weight", |
| "shape": [ |
| 16384, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 16777216, |
| "byteOffset": 11800576 |
| }, |
| { |
| "name": "model.layers.7.mlp.gate_up_proj.q_scale", |
| "shape": [ |
| 16384, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 28577792 |
| }, |
| { |
| "name": "model.layers.7.post_attention_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 30674944 |
| } |
| ], |
| "md5sum": "ad8d370c9603a4cb7a23e17b3d8df3b7" |
| }, |
| { |
| "dataPath": "params_shard_19.bin", |
| "format": "raw-shard", |
| "nbytes": 32116736, |
| "records": [ |
| { |
| "name": "model.layers.7.self_attn.qkv_proj.q_weight", |
| "shape": [ |
| 3072, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 3145728, |
| "byteOffset": 0 |
| }, |
| { |
| "name": "model.layers.7.self_attn.qkv_proj.q_scale", |
| "shape": [ |
| 3072, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 393216, |
| "byteOffset": 3145728 |
| }, |
| { |
| "name": "model.layers.7.self_attn.o_proj.q_weight", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 3538944 |
| }, |
| { |
| "name": "model.layers.7.self_attn.o_proj.q_scale", |
| "shape": [ |
| 2048, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 262144, |
| "byteOffset": 5636096 |
| }, |
| { |
| "name": "model.layers.8.input_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 5898240 |
| }, |
| { |
| "name": "model.layers.8.mlp.down_proj.q_weight", |
| "shape": [ |
| 2048, |
| 1024 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 8388608, |
| "byteOffset": 5902336 |
| }, |
| { |
| "name": "model.layers.8.mlp.down_proj.q_scale", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 1048576, |
| "byteOffset": 14290944 |
| }, |
| { |
| "name": "model.layers.8.mlp.gate_up_proj.q_weight", |
| "shape": [ |
| 16384, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 16777216, |
| "byteOffset": 15339520 |
| } |
| ], |
| "md5sum": "23b562f2a370825fae321c560bb522b8" |
| }, |
| { |
| "dataPath": "params_shard_20.bin", |
| "format": "raw-shard", |
| "nbytes": 16777216, |
| "records": [ |
| { |
| "name": "model.layers.9.mlp.gate_up_proj.q_weight", |
| "shape": [ |
| 16384, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 16777216, |
| "byteOffset": 0 |
| } |
| ], |
| "md5sum": "d4f7772740d5b196b4edb2389b91bfc1" |
| }, |
| { |
| "dataPath": "params_shard_21.bin", |
| "format": "raw-shard", |
| "nbytes": 25444352, |
| "records": [ |
| { |
| "name": "model.layers.8.mlp.gate_up_proj.q_scale", |
| "shape": [ |
| 16384, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 0 |
| }, |
| { |
| "name": "model.layers.8.post_attention_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 2097152 |
| }, |
| { |
| "name": "model.layers.8.self_attn.qkv_proj.q_weight", |
| "shape": [ |
| 3072, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 3145728, |
| "byteOffset": 2101248 |
| }, |
| { |
| "name": "model.layers.8.self_attn.qkv_proj.q_scale", |
| "shape": [ |
| 3072, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 393216, |
| "byteOffset": 5246976 |
| }, |
| { |
| "name": "model.layers.8.self_attn.o_proj.q_weight", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 5640192 |
| }, |
| { |
| "name": "model.layers.8.self_attn.o_proj.q_scale", |
| "shape": [ |
| 2048, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 262144, |
| "byteOffset": 7737344 |
| }, |
| { |
| "name": "model.layers.9.input_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 7999488 |
| }, |
| { |
| "name": "model.layers.9.mlp.down_proj.q_weight", |
| "shape": [ |
| 2048, |
| 1024 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 8388608, |
| "byteOffset": 8003584 |
| }, |
| { |
| "name": "model.layers.9.mlp.down_proj.q_scale", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 1048576, |
| "byteOffset": 16392192 |
| }, |
| { |
| "name": "model.layers.9.mlp.gate_up_proj.q_scale", |
| "shape": [ |
| 16384, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 17440768 |
| }, |
| { |
| "name": "model.layers.9.post_attention_layernorm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 19537920 |
| }, |
| { |
| "name": "model.layers.9.self_attn.qkv_proj.q_weight", |
| "shape": [ |
| 3072, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 3145728, |
| "byteOffset": 19542016 |
| }, |
| { |
| "name": "model.layers.9.self_attn.qkv_proj.q_scale", |
| "shape": [ |
| 3072, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 393216, |
| "byteOffset": 22687744 |
| }, |
| { |
| "name": "model.layers.9.self_attn.o_proj.q_weight", |
| "shape": [ |
| 2048, |
| 256 |
| ], |
| "dtype": "uint32", |
| "format": "f32-to-bf16", |
| "nbytes": 2097152, |
| "byteOffset": 23080960 |
| }, |
| { |
| "name": "model.layers.9.self_attn.o_proj.q_scale", |
| "shape": [ |
| 2048, |
| 64 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 262144, |
| "byteOffset": 25178112 |
| }, |
| { |
| "name": "model.norm.weight", |
| "shape": [ |
| 2048 |
| ], |
| "dtype": "float16", |
| "format": "f32-to-bf16", |
| "nbytes": 4096, |
| "byteOffset": 25440256 |
| } |
| ], |
| "md5sum": "af26897b30f986377a5cb178c8b1c09f" |
| } |
| ] |
| } |