| { | |
| "metadata": { | |
| "ParamSize": 195, | |
| "ParamBytes": 16061046784.0, | |
| "BitsPerParam": 16.0 | |
| }, | |
| "records": [ | |
| { | |
| "dataPath": "params_shard_0.bin", | |
| "format": "raw-shard", | |
| "nbytes": 1050935296, | |
| "records": [ | |
| { | |
| "name": "lm_head.weight", | |
| "shape": [ | |
| 128288, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 1050935296, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "ecc06a096cc46370a911d269bbd5e89e" | |
| }, | |
| { | |
| "dataPath": "params_shard_1.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.31.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "ee4a154632d2aa6ec82695ddbafbfb63" | |
| }, | |
| { | |
| "dataPath": "params_shard_2.bin", | |
| "format": "raw-shard", | |
| "nbytes": 1050935296, | |
| "records": [ | |
| { | |
| "name": "model.embed_tokens.weight", | |
| "shape": [ | |
| 128288, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 1050935296, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "9b55ef82466a0cecb8803f3e967983fe" | |
| }, | |
| { | |
| "dataPath": "params_shard_3.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.0.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "e299cfdb94c63c55290ad929e4eca8c1" | |
| }, | |
| { | |
| "dataPath": "params_shard_4.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.0.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "d8cdcb042f7a39d94713b887d5b916f1" | |
| }, | |
| { | |
| "dataPath": "params_shard_5.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.0.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "6b53591e6806913e62194d0aaddf1c55" | |
| }, | |
| { | |
| "dataPath": "params_shard_6.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.0.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "bff5bf7cf6f2b9e38424890fad97941a" | |
| }, | |
| { | |
| "dataPath": "params_shard_7.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.1.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "d3802573bac229e683f051cedc46e5c3" | |
| }, | |
| { | |
| "dataPath": "params_shard_8.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.1.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "d80c6661159286f5216ffdb675171fe1" | |
| }, | |
| { | |
| "dataPath": "params_shard_9.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.1.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "0d4f59e188d783de0996609e0db405b4" | |
| }, | |
| { | |
| "dataPath": "params_shard_10.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.1.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "294d15843280ddcd89921f3963ae0ced" | |
| }, | |
| { | |
| "dataPath": "params_shard_11.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.2.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "4e428b66993d59889e5e00017d310f74" | |
| }, | |
| { | |
| "dataPath": "params_shard_12.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.2.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "3c0159be15cf1b57584b8e089aa59f2d" | |
| }, | |
| { | |
| "dataPath": "params_shard_13.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.2.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "da7cfecb809cf4df1260c5320316fb41" | |
| }, | |
| { | |
| "dataPath": "params_shard_14.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.2.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "94f165b68936c1c9d6262cb089ddf235" | |
| }, | |
| { | |
| "dataPath": "params_shard_15.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.3.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "3368d8027f0576822f7df5af8169d053" | |
| }, | |
| { | |
| "dataPath": "params_shard_16.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.3.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "711364dfae36037bb310a9c8cecc2c74" | |
| }, | |
| { | |
| "dataPath": "params_shard_17.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.3.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "24f125b60fe99a9f293fba646d0ebd42" | |
| }, | |
| { | |
| "dataPath": "params_shard_18.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.3.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "79e23f3d7ce6c26b6c88a04af6ed82fc" | |
| }, | |
| { | |
| "dataPath": "params_shard_19.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.4.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "f344c3acbe3da422b407c3adcc5b2d29" | |
| }, | |
| { | |
| "dataPath": "params_shard_20.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.4.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "3703bbd20d99ff0897a42d777ed560fe" | |
| }, | |
| { | |
| "dataPath": "params_shard_21.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.4.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "2e5b2055569e8710e397d157c6d796cc" | |
| }, | |
| { | |
| "dataPath": "params_shard_22.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.4.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "d6b16914e4de0e6c54a9e4e9153f7493" | |
| }, | |
| { | |
| "dataPath": "params_shard_23.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.5.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "fd59471065f91a06c80c3b5b6def68ef" | |
| }, | |
| { | |
| "dataPath": "params_shard_24.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.5.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "634b5527134e6e49d0d86fa0508228c5" | |
| }, | |
| { | |
| "dataPath": "params_shard_25.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.5.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "3c78b36eed963699eb885366d64dbd96" | |
| }, | |
| { | |
| "dataPath": "params_shard_26.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.5.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "b9f5620c225ba265d437cb12ef5e4532" | |
| }, | |
| { | |
| "dataPath": "params_shard_27.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.6.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "07acdc287c5f3a993c43c56b50f8a57d" | |
| }, | |
| { | |
| "dataPath": "params_shard_28.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.6.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "318780da8705700eec67eb29285a59b5" | |
| }, | |
| { | |
| "dataPath": "params_shard_29.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.6.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "9f3255ebb7004d9aae3f787f0e6927ed" | |
| }, | |
| { | |
| "dataPath": "params_shard_30.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.6.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "ec3f792c415e82cc0a8420165ed2164f" | |
| }, | |
| { | |
| "dataPath": "params_shard_31.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.7.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "65e9d34285f1acc4845690779d4cb420" | |
| }, | |
| { | |
| "dataPath": "params_shard_32.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.7.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "b9c7303f9ebf7172eaea3dfb2a38956f" | |
| }, | |
| { | |
| "dataPath": "params_shard_33.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.7.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "c7d1135b98a1406ea6bde227cc018337" | |
| }, | |
| { | |
| "dataPath": "params_shard_34.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.7.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "bf67a5755b8da8c2ca8db9e93b456a7f" | |
| }, | |
| { | |
| "dataPath": "params_shard_35.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.8.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "2810b4f0b5d573cd47cec1e86c7442b3" | |
| }, | |
| { | |
| "dataPath": "params_shard_36.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.8.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "90095a9f1d1b61e5eca59307e21f5522" | |
| }, | |
| { | |
| "dataPath": "params_shard_37.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.8.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "3b69717b3068aa2d5fbe4302e0c30ae1" | |
| }, | |
| { | |
| "dataPath": "params_shard_38.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.8.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "09da3922c3d5bcb28b4c7ae1cd6642c0" | |
| }, | |
| { | |
| "dataPath": "params_shard_39.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.10.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "0fbeb9e99ef80514bfa46dbc609900b4" | |
| }, | |
| { | |
| "dataPath": "params_shard_40.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.10.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "5bc392f7980a15716bb170096095315e" | |
| }, | |
| { | |
| "dataPath": "params_shard_41.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.10.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "3320680dc8ea2d0e6cd76d5c9aa24fb7" | |
| }, | |
| { | |
| "dataPath": "params_shard_42.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.10.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "952afef482d6570924e9d23b2263571c" | |
| }, | |
| { | |
| "dataPath": "params_shard_43.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.11.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "92d254453e9626504f2ab1519cb1c224" | |
| }, | |
| { | |
| "dataPath": "params_shard_44.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.11.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "8b1d3e60c801e1ee7c61699b12a296e7" | |
| }, | |
| { | |
| "dataPath": "params_shard_45.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.11.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "f53c07f0ed94b3b97341fd61391532d7" | |
| }, | |
| { | |
| "dataPath": "params_shard_46.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.11.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "16e0b92f45eb1d0f4efafb264255e686" | |
| }, | |
| { | |
| "dataPath": "params_shard_47.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.12.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "d688acd969d50dfd1684a380420f60ab" | |
| }, | |
| { | |
| "dataPath": "params_shard_48.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.12.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "82e1c71cde9d5c4df1264e1683b47262" | |
| }, | |
| { | |
| "dataPath": "params_shard_49.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.12.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "0e273234aa85c24c21f2e6cb0ad03895" | |
| }, | |
| { | |
| "dataPath": "params_shard_50.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.12.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "e99c358b4bd4ecdb0ca84b2f9b3d74a4" | |
| }, | |
| { | |
| "dataPath": "params_shard_51.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.13.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "cc86dac581bdfcd639ddd6bbe9140c65" | |
| }, | |
| { | |
| "dataPath": "params_shard_52.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.13.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "46600ec8024bfe1ce767d97c4a69f688" | |
| }, | |
| { | |
| "dataPath": "params_shard_53.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.13.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "4f2269d428f9e510bc054de7fb569909" | |
| }, | |
| { | |
| "dataPath": "params_shard_54.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.13.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "849fc496126efc64ac8ab0a561a3e948" | |
| }, | |
| { | |
| "dataPath": "params_shard_55.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.14.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "1ec4b31b9d23cf97335406d97d746ddc" | |
| }, | |
| { | |
| "dataPath": "params_shard_56.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.14.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "bd4d6f857899366395c687b3ed9a0bff" | |
| }, | |
| { | |
| "dataPath": "params_shard_57.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.14.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "529ace6a5c4aade2f65fa479693b009d" | |
| }, | |
| { | |
| "dataPath": "params_shard_58.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.14.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "4620cbfd87a9e07572274916ea981a18" | |
| }, | |
| { | |
| "dataPath": "params_shard_59.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.15.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "6199f83db41c1128e0c9cd8e1e977765" | |
| }, | |
| { | |
| "dataPath": "params_shard_60.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.15.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "512ef8e28305e5e5e166f8deeae9fd75" | |
| }, | |
| { | |
| "dataPath": "params_shard_61.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.15.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "14c498441e1b38758f3c22531d99770e" | |
| }, | |
| { | |
| "dataPath": "params_shard_62.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.15.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "01cc2e64847e4fa3c913a5ebb29dee29" | |
| }, | |
| { | |
| "dataPath": "params_shard_63.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.16.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "200463564c6f43fdb9c59929fb1b002f" | |
| }, | |
| { | |
| "dataPath": "params_shard_64.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.16.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "900e52fb23be63e911270d93f6ad8df0" | |
| }, | |
| { | |
| "dataPath": "params_shard_65.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.16.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "416d6fccab1046af0c77a037fd85ecd9" | |
| }, | |
| { | |
| "dataPath": "params_shard_66.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.16.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "5bdfb454c86fbf867f0819cbb8078d1a" | |
| }, | |
| { | |
| "dataPath": "params_shard_67.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.17.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "470d14b53d09559159cfd374ac1f7c70" | |
| }, | |
| { | |
| "dataPath": "params_shard_68.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.17.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "1765a6baeaec42344df4243a20816557" | |
| }, | |
| { | |
| "dataPath": "params_shard_69.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.17.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "a7e6ce1d1e21fdf6ff9d53e11f76cc68" | |
| }, | |
| { | |
| "dataPath": "params_shard_70.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.17.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "37bc2abebdb254018c4f790644df56b7" | |
| }, | |
| { | |
| "dataPath": "params_shard_71.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.18.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "9b28bb0a46827ef741166a37628913df" | |
| }, | |
| { | |
| "dataPath": "params_shard_72.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.18.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "c1d278e1cb4b34cb63bf13996c929b92" | |
| }, | |
| { | |
| "dataPath": "params_shard_73.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.18.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "38263d822dcfd1fb390194ce0c2afbbe" | |
| }, | |
| { | |
| "dataPath": "params_shard_74.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.18.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "c1889c49ac58b2374598ac44f27fc402" | |
| }, | |
| { | |
| "dataPath": "params_shard_75.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.19.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "5ab5fe752e4e7a3d1bbc4c2790c03c3e" | |
| }, | |
| { | |
| "dataPath": "params_shard_76.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.19.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "317de9668b83b33f71a0df5ed706462b" | |
| }, | |
| { | |
| "dataPath": "params_shard_77.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.19.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "f216b1918b31b70188813190663fccf5" | |
| }, | |
| { | |
| "dataPath": "params_shard_78.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.19.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "aec24dfbf87691bde19219d16e1931de" | |
| }, | |
| { | |
| "dataPath": "params_shard_79.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.20.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "63d47b3c72b5020e49f563954a1b8faf" | |
| }, | |
| { | |
| "dataPath": "params_shard_80.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.20.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "8dee9526410ec70519d1e324a1d60667" | |
| }, | |
| { | |
| "dataPath": "params_shard_81.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.20.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "42c6c5ce2cf5a9b83e64e48d87b37629" | |
| }, | |
| { | |
| "dataPath": "params_shard_82.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.9.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "61cdd1b2151ed4b40634fe048f188496" | |
| }, | |
| { | |
| "dataPath": "params_shard_83.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.9.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "4b83a302de3fa153ae806c774b137399" | |
| }, | |
| { | |
| "dataPath": "params_shard_84.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.9.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "00e5d3319304e419f6c09ad818e12a77" | |
| }, | |
| { | |
| "dataPath": "params_shard_85.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.9.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "8b363d3bc2b6e8d79a696cb96e2fc7b2" | |
| }, | |
| { | |
| "dataPath": "params_shard_86.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.20.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "07d739b4c2a0f3bfa6652a10de792b4c" | |
| }, | |
| { | |
| "dataPath": "params_shard_87.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.21.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "4fd9c64fca40b7a50a475347199441f4" | |
| }, | |
| { | |
| "dataPath": "params_shard_88.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.21.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "c84bca616c4a9688bd63025c8b0d81d8" | |
| }, | |
| { | |
| "dataPath": "params_shard_89.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.21.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "9c8d7f792747c0f340b28dee0b6a24e0" | |
| }, | |
| { | |
| "dataPath": "params_shard_90.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.21.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "2faa78d25f3d1059a4bd4b152d6c58b1" | |
| }, | |
| { | |
| "dataPath": "params_shard_91.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.22.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "a6be7d4ab4a12716cb5d797c8d97b23e" | |
| }, | |
| { | |
| "dataPath": "params_shard_92.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.22.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "158af4361381bbc3c1e02f0ca72fa557" | |
| }, | |
| { | |
| "dataPath": "params_shard_93.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.22.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "a06111be4178b8872a0b3df3a7b41ee5" | |
| }, | |
| { | |
| "dataPath": "params_shard_94.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.22.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "d0dc953859b24377fa06a9f40d68f112" | |
| }, | |
| { | |
| "dataPath": "params_shard_95.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.23.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "9fcc3dd5c9b4829f51e917ff8b738c35" | |
| }, | |
| { | |
| "dataPath": "params_shard_96.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.23.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "d9f34f85144950f1b68de11465101a63" | |
| }, | |
| { | |
| "dataPath": "params_shard_97.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.23.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "5dad47f20c2f55b06522e87bdea92fe0" | |
| }, | |
| { | |
| "dataPath": "params_shard_98.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.23.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "269cd1065d8c768ed698df343dab67f2" | |
| }, | |
| { | |
| "dataPath": "params_shard_99.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.24.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "17232eb4610869341d76db804df3f642" | |
| }, | |
| { | |
| "dataPath": "params_shard_100.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.24.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "ce7fbe456bbbc56154dd77d3e8ff08c2" | |
| }, | |
| { | |
| "dataPath": "params_shard_101.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.24.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "896b0cb56fd740651d3cf3f9dea2e455" | |
| }, | |
| { | |
| "dataPath": "params_shard_102.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.24.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "080bd372b094619404fed9ebdf4bcdd4" | |
| }, | |
| { | |
| "dataPath": "params_shard_103.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.25.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "21bf5aa0bec91cfc18959c41b77c282c" | |
| }, | |
| { | |
| "dataPath": "params_shard_104.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.25.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "4a86cf6bac26ce4d14a0112e5191df74" | |
| }, | |
| { | |
| "dataPath": "params_shard_105.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.25.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "5df276d49abd874fcf939e359aef821b" | |
| }, | |
| { | |
| "dataPath": "params_shard_106.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.25.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "d415c21d2b63f1e3518bfb75589d38fe" | |
| }, | |
| { | |
| "dataPath": "params_shard_107.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.26.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "eb66a11880da238253920f4a068d26f3" | |
| }, | |
| { | |
| "dataPath": "params_shard_108.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.26.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "81854121c17153449410339970187a7b" | |
| }, | |
| { | |
| "dataPath": "params_shard_109.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.26.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "10b242a7f52cdbdae52f1b7621a1399e" | |
| }, | |
| { | |
| "dataPath": "params_shard_110.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.26.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "4281af70e081af3642e33c7e9621fe4a" | |
| }, | |
| { | |
| "dataPath": "params_shard_111.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.27.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "d42592c61a1695aa500dcbe3f79d1812" | |
| }, | |
| { | |
| "dataPath": "params_shard_112.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.27.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "e7d47b710cf5f7b01bb7fcd18c206530" | |
| }, | |
| { | |
| "dataPath": "params_shard_113.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.27.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "887f350e05a267120492050fdf02e15a" | |
| }, | |
| { | |
| "dataPath": "params_shard_114.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.27.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "792ef99016509010d7f32daab347be17" | |
| }, | |
| { | |
| "dataPath": "params_shard_115.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.28.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "24f9dce1b3068c804a89f2024488fd72" | |
| }, | |
| { | |
| "dataPath": "params_shard_116.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.28.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "9b548580ec4b987755519552cc399639" | |
| }, | |
| { | |
| "dataPath": "params_shard_117.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.28.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "030e5ce73825551d49ca755a8df6f446" | |
| }, | |
| { | |
| "dataPath": "params_shard_118.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.28.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "bbd00e86746f5c8c6bb65f319903769a" | |
| }, | |
| { | |
| "dataPath": "params_shard_119.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.29.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "16746644ed74a86f83c89dc7c6d821ce" | |
| }, | |
| { | |
| "dataPath": "params_shard_120.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.29.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "b1bc622966bcbeb05dbf782f949c693e" | |
| }, | |
| { | |
| "dataPath": "params_shard_121.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.29.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "77542cd9bde7d1d7cbe94d6b0ae26a82" | |
| }, | |
| { | |
| "dataPath": "params_shard_122.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.29.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "2217ae7b02d6bd41d5f5a901f0db6d95" | |
| }, | |
| { | |
| "dataPath": "params_shard_123.bin", | |
| "format": "raw-shard", | |
| "nbytes": 117440512, | |
| "records": [ | |
| { | |
| "name": "model.layers.30.mlp.down_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 14336 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 117440512, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "eb083bebc505b1df45fc7c99e4b53fca" | |
| }, | |
| { | |
| "dataPath": "params_shard_124.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.30.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "b99bc95577936d7447a2cd4cc4335638" | |
| }, | |
| { | |
| "dataPath": "params_shard_125.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.30.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "ad4d020a19b0ad8f6cce7d662eb7858b" | |
| }, | |
| { | |
| "dataPath": "params_shard_126.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.30.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "c4eaba9cd647b01b529b4a712e1c8987" | |
| }, | |
| { | |
| "dataPath": "params_shard_127.bin", | |
| "format": "raw-shard", | |
| "nbytes": 234881024, | |
| "records": [ | |
| { | |
| "name": "model.layers.31.mlp.gate_up_proj.weight", | |
| "shape": [ | |
| 28672, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 234881024, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "2b6472f4e43db59028df41b27e2c3345" | |
| }, | |
| { | |
| "dataPath": "params_shard_128.bin", | |
| "format": "raw-shard", | |
| "nbytes": 50331648, | |
| "records": [ | |
| { | |
| "name": "model.layers.31.self_attn.qkv_proj.weight", | |
| "shape": [ | |
| 6144, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 50331648, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "65442520600fc32cd2ad711c9f9b2db5" | |
| }, | |
| { | |
| "dataPath": "params_shard_129.bin", | |
| "format": "raw-shard", | |
| "nbytes": 33554432, | |
| "records": [ | |
| { | |
| "name": "model.layers.31.self_attn.o_proj.weight", | |
| "shape": [ | |
| 4096, | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 33554432, | |
| "byteOffset": 0 | |
| } | |
| ], | |
| "md5sum": "d0a869bdd171ae88d7c4e0da62e9323f" | |
| }, | |
| { | |
| "dataPath": "params_shard_130.bin", | |
| "format": "raw-shard", | |
| "nbytes": 532480, | |
| "records": [ | |
| { | |
| "name": "model.layers.31.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 0 | |
| }, | |
| { | |
| "name": "model.layers.31.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 8192 | |
| }, | |
| { | |
| "name": "model.norm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 16384 | |
| }, | |
| { | |
| "name": "model.layers.0.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 24576 | |
| }, | |
| { | |
| "name": "model.layers.0.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 32768 | |
| }, | |
| { | |
| "name": "model.layers.1.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 40960 | |
| }, | |
| { | |
| "name": "model.layers.1.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 49152 | |
| }, | |
| { | |
| "name": "model.layers.2.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 57344 | |
| }, | |
| { | |
| "name": "model.layers.2.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 65536 | |
| }, | |
| { | |
| "name": "model.layers.3.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 73728 | |
| }, | |
| { | |
| "name": "model.layers.3.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 81920 | |
| }, | |
| { | |
| "name": "model.layers.4.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 90112 | |
| }, | |
| { | |
| "name": "model.layers.4.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 98304 | |
| }, | |
| { | |
| "name": "model.layers.5.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 106496 | |
| }, | |
| { | |
| "name": "model.layers.5.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 114688 | |
| }, | |
| { | |
| "name": "model.layers.6.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 122880 | |
| }, | |
| { | |
| "name": "model.layers.6.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 131072 | |
| }, | |
| { | |
| "name": "model.layers.7.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 139264 | |
| }, | |
| { | |
| "name": "model.layers.7.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 147456 | |
| }, | |
| { | |
| "name": "model.layers.8.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 155648 | |
| }, | |
| { | |
| "name": "model.layers.8.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 163840 | |
| }, | |
| { | |
| "name": "model.layers.10.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 172032 | |
| }, | |
| { | |
| "name": "model.layers.10.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 180224 | |
| }, | |
| { | |
| "name": "model.layers.11.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 188416 | |
| }, | |
| { | |
| "name": "model.layers.11.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 196608 | |
| }, | |
| { | |
| "name": "model.layers.12.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 204800 | |
| }, | |
| { | |
| "name": "model.layers.12.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 212992 | |
| }, | |
| { | |
| "name": "model.layers.13.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 221184 | |
| }, | |
| { | |
| "name": "model.layers.13.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 229376 | |
| }, | |
| { | |
| "name": "model.layers.14.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 237568 | |
| }, | |
| { | |
| "name": "model.layers.14.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 245760 | |
| }, | |
| { | |
| "name": "model.layers.15.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 253952 | |
| }, | |
| { | |
| "name": "model.layers.15.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 262144 | |
| }, | |
| { | |
| "name": "model.layers.16.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 270336 | |
| }, | |
| { | |
| "name": "model.layers.16.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 278528 | |
| }, | |
| { | |
| "name": "model.layers.17.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 286720 | |
| }, | |
| { | |
| "name": "model.layers.17.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 294912 | |
| }, | |
| { | |
| "name": "model.layers.18.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 303104 | |
| }, | |
| { | |
| "name": "model.layers.18.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 311296 | |
| }, | |
| { | |
| "name": "model.layers.19.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 319488 | |
| }, | |
| { | |
| "name": "model.layers.19.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 327680 | |
| }, | |
| { | |
| "name": "model.layers.9.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 335872 | |
| }, | |
| { | |
| "name": "model.layers.9.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 344064 | |
| }, | |
| { | |
| "name": "model.layers.20.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 352256 | |
| }, | |
| { | |
| "name": "model.layers.20.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 360448 | |
| }, | |
| { | |
| "name": "model.layers.21.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 368640 | |
| }, | |
| { | |
| "name": "model.layers.21.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 376832 | |
| }, | |
| { | |
| "name": "model.layers.22.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 385024 | |
| }, | |
| { | |
| "name": "model.layers.22.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 393216 | |
| }, | |
| { | |
| "name": "model.layers.23.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 401408 | |
| }, | |
| { | |
| "name": "model.layers.23.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 409600 | |
| }, | |
| { | |
| "name": "model.layers.24.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 417792 | |
| }, | |
| { | |
| "name": "model.layers.24.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 425984 | |
| }, | |
| { | |
| "name": "model.layers.25.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 434176 | |
| }, | |
| { | |
| "name": "model.layers.25.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 442368 | |
| }, | |
| { | |
| "name": "model.layers.26.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 450560 | |
| }, | |
| { | |
| "name": "model.layers.26.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 458752 | |
| }, | |
| { | |
| "name": "model.layers.27.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 466944 | |
| }, | |
| { | |
| "name": "model.layers.27.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 475136 | |
| }, | |
| { | |
| "name": "model.layers.28.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 483328 | |
| }, | |
| { | |
| "name": "model.layers.28.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 491520 | |
| }, | |
| { | |
| "name": "model.layers.29.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 499712 | |
| }, | |
| { | |
| "name": "model.layers.29.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 507904 | |
| }, | |
| { | |
| "name": "model.layers.30.input_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 516096 | |
| }, | |
| { | |
| "name": "model.layers.30.post_attention_layernorm.weight", | |
| "shape": [ | |
| 4096 | |
| ], | |
| "dtype": "float16", | |
| "format": "f32-to-bf16", | |
| "nbytes": 8192, | |
| "byteOffset": 524288 | |
| } | |
| ], | |
| "md5sum": "29d44acb4890cf05806a9cef56b982ac" | |
| } | |
| ] | |
| } |