{ "metadata": { "ParamSize": 518, "ParamBytes": 1734430720.0, "BitsPerParam": 5.010432978788311 }, "records": [ { "dataPath": "params_shard_0.bin", "format": "raw-shard", "nbytes": 62914560, "records": [ { "name": "gpt_neox.embed_in.q_weight", "shape": [ 49152, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 62914560, "byteOffset": 0 } ], "md5sum": "130d0e0e3d1fa08301f4c0a224eb2a93" }, { "dataPath": "params_shard_1.bin", "format": "raw-shard", "nbytes": 22650880, "records": [ { "name": "gpt_neox.embed_in.q_scale", "shape": [ 49152, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 0 }, { "name": "gpt_neox.layers.0.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 7864320 }, { "name": "gpt_neox.layers.0.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 7869440 }, { "name": "gpt_neox.layers.0.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 7874560 }, { "name": "gpt_neox.layers.0.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 7879680 }, { "name": "gpt_neox.layers.0.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 7884800 }, { "name": "gpt_neox.layers.0.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 17715200 }, { "name": "gpt_neox.layers.0.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 18944000 }, { "name": "gpt_neox.layers.0.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 18959360 }, { "name": "gpt_neox.layers.0.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 22236160 }, { "name": "gpt_neox.layers.0.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 22645760 } ], "md5sum": "18099736e1a0ff25881880e0accb4989" }, { "dataPath": "params_shard_2.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.0.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.0.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.0.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.0.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.0.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.0.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29511680 }, { "name": "gpt_neox.layers.1.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29516800 }, { "name": "gpt_neox.layers.1.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29521920 }, { "name": "gpt_neox.layers.1.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29527040 }, { "name": "gpt_neox.layers.1.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "23f6fc332f115869ac83adbd80a3f200" }, { "dataPath": "params_shard_3.bin", "format": "raw-shard", "nbytes": 29532160, "records": [ { "name": "gpt_neox.layers.1.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "gpt_neox.layers.1.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 9830400 }, { "name": "gpt_neox.layers.1.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 11059200 }, { "name": "gpt_neox.layers.1.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 11074560 }, { "name": "gpt_neox.layers.1.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 14351360 }, { "name": "gpt_neox.layers.1.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.1.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.1.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.1.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 29511680 } ], "md5sum": "67d21fff48aa7f2a71f5412b16d16e1f" }, { "dataPath": "params_shard_4.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.1.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.1.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.1.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.2.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14750720 }, { "name": "gpt_neox.layers.2.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14755840 }, { "name": "gpt_neox.layers.2.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.2.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.2.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 14771200 }, { "name": "gpt_neox.layers.2.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 24601600 }, { "name": "gpt_neox.layers.2.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 25830400 }, { "name": "gpt_neox.layers.2.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 25845760 }, { "name": "gpt_neox.layers.2.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 29122560 }, { "name": "gpt_neox.layers.2.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "83ccd56fcf9a6c83228308f5116c200f" }, { "dataPath": "params_shard_5.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.2.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.2.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.2.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.2.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.2.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.2.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29511680 }, { "name": "gpt_neox.layers.3.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29516800 }, { "name": "gpt_neox.layers.3.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29521920 }, { "name": "gpt_neox.layers.3.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29527040 }, { "name": "gpt_neox.layers.3.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "b024119fb95076ba55ab9be41f09ed96" }, { "dataPath": "params_shard_6.bin", "format": "raw-shard", "nbytes": 29532160, "records": [ { "name": "gpt_neox.layers.3.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "gpt_neox.layers.3.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 9830400 }, { "name": "gpt_neox.layers.3.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 11059200 }, { "name": "gpt_neox.layers.3.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 11074560 }, { "name": "gpt_neox.layers.3.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 14351360 }, { "name": "gpt_neox.layers.3.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.3.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.3.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.3.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 29511680 } ], "md5sum": "d16bede9469ef0d657644d7b0b874953" }, { "dataPath": "params_shard_7.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.3.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.3.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.3.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.4.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14750720 }, { "name": "gpt_neox.layers.4.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14755840 }, { "name": "gpt_neox.layers.4.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.4.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.4.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 14771200 }, { "name": "gpt_neox.layers.4.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 24601600 }, { "name": "gpt_neox.layers.4.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 25830400 }, { "name": "gpt_neox.layers.4.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 25845760 }, { "name": "gpt_neox.layers.4.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 29122560 }, { "name": "gpt_neox.layers.4.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "0273c2053de2bbba3c8addab8b51f29e" }, { "dataPath": "params_shard_8.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.4.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.4.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.4.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.4.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.4.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.4.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29511680 }, { "name": "gpt_neox.layers.5.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29516800 }, { "name": "gpt_neox.layers.5.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29521920 }, { "name": "gpt_neox.layers.5.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29527040 }, { "name": "gpt_neox.layers.5.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "c66d4d233e16e5564c8eb78db61a0ab0" }, { "dataPath": "params_shard_9.bin", "format": "raw-shard", "nbytes": 29532160, "records": [ { "name": "gpt_neox.layers.5.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "gpt_neox.layers.5.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 9830400 }, { "name": "gpt_neox.layers.5.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 11059200 }, { "name": "gpt_neox.layers.5.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 11074560 }, { "name": "gpt_neox.layers.5.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 14351360 }, { "name": "gpt_neox.layers.5.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.5.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.5.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.5.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 29511680 } ], "md5sum": "6168569092ea967a6b0bc1d667eaabe5" }, { "dataPath": "params_shard_10.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.5.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.5.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.5.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.6.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14750720 }, { "name": "gpt_neox.layers.6.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14755840 }, { "name": "gpt_neox.layers.6.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.6.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.6.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 14771200 }, { "name": "gpt_neox.layers.6.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 24601600 }, { "name": "gpt_neox.layers.6.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 25830400 }, { "name": "gpt_neox.layers.6.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 25845760 }, { "name": "gpt_neox.layers.6.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 29122560 }, { "name": "gpt_neox.layers.6.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "61d629151f9cf05ba8a6b9f156e91de5" }, { "dataPath": "params_shard_11.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.6.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.6.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.6.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.6.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.6.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.6.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29511680 }, { "name": "gpt_neox.layers.7.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29516800 }, { "name": "gpt_neox.layers.7.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29521920 }, { "name": "gpt_neox.layers.7.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29527040 }, { "name": "gpt_neox.layers.7.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "89e49c725e0aa441eef082b8aa282010" }, { "dataPath": "params_shard_12.bin", "format": "raw-shard", "nbytes": 29532160, "records": [ { "name": "gpt_neox.layers.7.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "gpt_neox.layers.7.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 9830400 }, { "name": "gpt_neox.layers.7.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 11059200 }, { "name": "gpt_neox.layers.7.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 11074560 }, { "name": "gpt_neox.layers.7.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 14351360 }, { "name": "gpt_neox.layers.7.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.7.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.7.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.7.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 29511680 } ], "md5sum": "269f41cc257440a6bf82f5b2dc8b60e2" }, { "dataPath": "params_shard_13.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.7.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.7.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.7.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.8.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14750720 }, { "name": "gpt_neox.layers.8.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14755840 }, { "name": "gpt_neox.layers.8.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.8.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.8.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 14771200 }, { "name": "gpt_neox.layers.8.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 24601600 }, { "name": "gpt_neox.layers.8.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 25830400 }, { "name": "gpt_neox.layers.8.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 25845760 }, { "name": "gpt_neox.layers.8.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 29122560 }, { "name": "gpt_neox.layers.8.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "bd24e6eb7b331cf9a28fdecadc179109" }, { "dataPath": "params_shard_14.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.8.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.8.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.8.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.8.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.8.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.8.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29511680 }, { "name": "gpt_neox.layers.9.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29516800 }, { "name": "gpt_neox.layers.9.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29521920 }, { "name": "gpt_neox.layers.9.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29527040 }, { "name": "gpt_neox.layers.9.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "6345876d27b4c166e3572ffe841159ad" }, { "dataPath": "params_shard_15.bin", "format": "raw-shard", "nbytes": 29532160, "records": [ { "name": "gpt_neox.layers.9.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "gpt_neox.layers.9.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 9830400 }, { "name": "gpt_neox.layers.9.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 11059200 }, { "name": "gpt_neox.layers.9.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 11074560 }, { "name": "gpt_neox.layers.9.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 14351360 }, { "name": "gpt_neox.layers.9.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.9.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.9.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.9.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 29511680 } ], "md5sum": "19c19ef98553f64b91f1bed9b5ef4032" }, { "dataPath": "params_shard_16.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.9.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.9.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.9.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.10.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14750720 }, { "name": "gpt_neox.layers.10.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14755840 }, { "name": "gpt_neox.layers.10.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.10.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.10.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 14771200 }, { "name": "gpt_neox.layers.10.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 24601600 }, { "name": "gpt_neox.layers.10.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 25830400 }, { "name": "gpt_neox.layers.10.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 25845760 }, { "name": "gpt_neox.layers.10.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 29122560 }, { "name": "gpt_neox.layers.10.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "c9e457cde4bd3fc22642f20cc4d19979" }, { "dataPath": "params_shard_17.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.10.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.10.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.10.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.10.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.10.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.10.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29511680 }, { "name": "gpt_neox.layers.11.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29516800 }, { "name": "gpt_neox.layers.11.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29521920 }, { "name": "gpt_neox.layers.11.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29527040 }, { "name": "gpt_neox.layers.11.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "af44068d1c581359eef7eb4f1080bb71" }, { "dataPath": "params_shard_18.bin", "format": "raw-shard", "nbytes": 29532160, "records": [ { "name": "gpt_neox.layers.11.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "gpt_neox.layers.11.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 9830400 }, { "name": "gpt_neox.layers.11.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 11059200 }, { "name": "gpt_neox.layers.11.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 11074560 }, { "name": "gpt_neox.layers.11.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 14351360 }, { "name": "gpt_neox.layers.11.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.11.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.11.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.11.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 29511680 } ], "md5sum": "7f7ae45a99e46394d123cfe3c07af799" }, { "dataPath": "params_shard_19.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.11.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.11.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.11.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.12.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14750720 }, { "name": "gpt_neox.layers.12.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14755840 }, { "name": "gpt_neox.layers.12.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.12.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.12.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 14771200 }, { "name": "gpt_neox.layers.12.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 24601600 }, { "name": "gpt_neox.layers.12.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 25830400 }, { "name": "gpt_neox.layers.12.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 25845760 }, { "name": "gpt_neox.layers.12.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 29122560 }, { "name": "gpt_neox.layers.12.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "17306d51500d13a9458adaa9a4d62826" }, { "dataPath": "params_shard_20.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.12.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.12.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.12.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.12.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.12.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.12.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29511680 }, { "name": "gpt_neox.layers.13.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29516800 }, { "name": "gpt_neox.layers.13.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29521920 }, { "name": "gpt_neox.layers.13.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29527040 }, { "name": "gpt_neox.layers.13.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "06b914336ed4f10684cbac356f7dfd5e" }, { "dataPath": "params_shard_21.bin", "format": "raw-shard", "nbytes": 29532160, "records": [ { "name": "gpt_neox.layers.13.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "gpt_neox.layers.13.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 9830400 }, { "name": "gpt_neox.layers.13.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 11059200 }, { "name": "gpt_neox.layers.13.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 11074560 }, { "name": "gpt_neox.layers.13.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 14351360 }, { "name": "gpt_neox.layers.13.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.13.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.13.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.13.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 29511680 } ], "md5sum": "6e8ca0dd1339cfc6cf7d0bd7d15c37fb" }, { "dataPath": "params_shard_22.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.13.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.13.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.13.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.14.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14750720 }, { "name": "gpt_neox.layers.14.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14755840 }, { "name": "gpt_neox.layers.14.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.14.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.14.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 14771200 }, { "name": "gpt_neox.layers.14.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 24601600 }, { "name": "gpt_neox.layers.14.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 25830400 }, { "name": "gpt_neox.layers.14.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 25845760 }, { "name": "gpt_neox.layers.14.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 29122560 }, { "name": "gpt_neox.layers.14.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "b53385ac21d45cbeaed1c341ac3e9eef" }, { "dataPath": "params_shard_23.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.14.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.14.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.14.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.14.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.14.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.14.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29511680 }, { "name": "gpt_neox.layers.15.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29516800 }, { "name": "gpt_neox.layers.15.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29521920 }, { "name": "gpt_neox.layers.15.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29527040 }, { "name": "gpt_neox.layers.15.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "64c15cf3bb3a4b225a26fe9d16e5b8e8" }, { "dataPath": "params_shard_24.bin", "format": "raw-shard", "nbytes": 29532160, "records": [ { "name": "gpt_neox.layers.15.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "gpt_neox.layers.15.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 9830400 }, { "name": "gpt_neox.layers.15.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 11059200 }, { "name": "gpt_neox.layers.15.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 11074560 }, { "name": "gpt_neox.layers.15.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 14351360 }, { "name": "gpt_neox.layers.15.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.15.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.15.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.15.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 29511680 } ], "md5sum": "3fed9efcb61a25f0973ba392889dd85e" }, { "dataPath": "params_shard_25.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.15.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.15.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.15.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.16.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14750720 }, { "name": "gpt_neox.layers.16.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14755840 }, { "name": "gpt_neox.layers.16.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.16.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.16.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 14771200 }, { "name": "gpt_neox.layers.16.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 24601600 }, { "name": "gpt_neox.layers.16.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 25830400 }, { "name": "gpt_neox.layers.16.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 25845760 }, { "name": "gpt_neox.layers.16.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 29122560 }, { "name": "gpt_neox.layers.16.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "7ccac867c6e382d0d542e8033ab3f775" }, { "dataPath": "params_shard_26.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.16.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.16.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.16.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.16.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.16.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.16.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29511680 }, { "name": "gpt_neox.layers.17.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29516800 }, { "name": "gpt_neox.layers.17.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29521920 }, { "name": "gpt_neox.layers.17.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29527040 }, { "name": "gpt_neox.layers.17.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "8bcd5c15ef2e875dfb94c4ebc1a209b3" }, { "dataPath": "params_shard_27.bin", "format": "raw-shard", "nbytes": 29532160, "records": [ { "name": "gpt_neox.layers.17.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "gpt_neox.layers.17.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 9830400 }, { "name": "gpt_neox.layers.17.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 11059200 }, { "name": "gpt_neox.layers.17.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 11074560 }, { "name": "gpt_neox.layers.17.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 14351360 }, { "name": "gpt_neox.layers.17.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.17.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.17.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.17.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 29511680 } ], "md5sum": "214c1183504e11834ad5572e1a179d7a" }, { "dataPath": "params_shard_28.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.17.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.17.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.17.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.18.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14750720 }, { "name": "gpt_neox.layers.18.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14755840 }, { "name": "gpt_neox.layers.18.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.18.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.18.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 14771200 }, { "name": "gpt_neox.layers.18.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 24601600 }, { "name": "gpt_neox.layers.18.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 25830400 }, { "name": "gpt_neox.layers.18.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 25845760 }, { "name": "gpt_neox.layers.18.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 29122560 }, { "name": "gpt_neox.layers.18.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "0f800a7329b9475d77a5d05f67cd5a5e" }, { "dataPath": "params_shard_29.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.18.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.18.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.18.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.18.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.18.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.18.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29511680 }, { "name": "gpt_neox.layers.19.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29516800 }, { "name": "gpt_neox.layers.19.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29521920 }, { "name": "gpt_neox.layers.19.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29527040 }, { "name": "gpt_neox.layers.19.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "0b34957f622e64e980722a305902a61a" }, { "dataPath": "params_shard_30.bin", "format": "raw-shard", "nbytes": 29532160, "records": [ { "name": "gpt_neox.layers.19.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "gpt_neox.layers.19.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 9830400 }, { "name": "gpt_neox.layers.19.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 11059200 }, { "name": "gpt_neox.layers.19.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 11074560 }, { "name": "gpt_neox.layers.19.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 14351360 }, { "name": "gpt_neox.layers.19.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.19.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.19.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.19.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 29511680 } ], "md5sum": "0ba73f9c34dabeb77088a8b06f62635a" }, { "dataPath": "params_shard_31.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.19.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.19.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.19.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.20.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14750720 }, { "name": "gpt_neox.layers.20.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14755840 }, { "name": "gpt_neox.layers.20.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.20.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.20.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 14771200 }, { "name": "gpt_neox.layers.20.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 24601600 }, { "name": "gpt_neox.layers.20.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 25830400 }, { "name": "gpt_neox.layers.20.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 25845760 }, { "name": "gpt_neox.layers.20.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 29122560 }, { "name": "gpt_neox.layers.20.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "77b6f858718a34d99d0d583dd2c3d2f6" }, { "dataPath": "params_shard_32.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.20.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.20.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.20.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.20.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.20.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.20.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29511680 }, { "name": "gpt_neox.layers.21.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29516800 }, { "name": "gpt_neox.layers.21.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29521920 }, { "name": "gpt_neox.layers.21.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29527040 }, { "name": "gpt_neox.layers.21.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "09216ea1e891008fc396c956c828131b" }, { "dataPath": "params_shard_33.bin", "format": "raw-shard", "nbytes": 29532160, "records": [ { "name": "gpt_neox.layers.21.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "gpt_neox.layers.21.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 9830400 }, { "name": "gpt_neox.layers.21.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 11059200 }, { "name": "gpt_neox.layers.21.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 11074560 }, { "name": "gpt_neox.layers.21.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 14351360 }, { "name": "gpt_neox.layers.21.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.21.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.21.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.21.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 29511680 } ], "md5sum": "91242d7567d250a96dd0e05c62ec787e" }, { "dataPath": "params_shard_34.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.21.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.21.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.21.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.22.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14750720 }, { "name": "gpt_neox.layers.22.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14755840 }, { "name": "gpt_neox.layers.22.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.22.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.22.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 14771200 }, { "name": "gpt_neox.layers.22.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 24601600 }, { "name": "gpt_neox.layers.22.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 25830400 }, { "name": "gpt_neox.layers.22.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 25845760 }, { "name": "gpt_neox.layers.22.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 29122560 }, { "name": "gpt_neox.layers.22.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "61f7cae329b8d1bdfed25bb7f13d262e" }, { "dataPath": "params_shard_35.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.22.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.22.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.22.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.22.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.22.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.22.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29511680 }, { "name": "gpt_neox.layers.23.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29516800 }, { "name": "gpt_neox.layers.23.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29521920 }, { "name": "gpt_neox.layers.23.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29527040 }, { "name": "gpt_neox.layers.23.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "3308bb6141a885c8cdd40d80092993ef" }, { "dataPath": "params_shard_36.bin", "format": "raw-shard", "nbytes": 29532160, "records": [ { "name": "gpt_neox.layers.23.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "gpt_neox.layers.23.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 9830400 }, { "name": "gpt_neox.layers.23.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 11059200 }, { "name": "gpt_neox.layers.23.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 11074560 }, { "name": "gpt_neox.layers.23.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 14351360 }, { "name": "gpt_neox.layers.23.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.23.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.23.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.23.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 29511680 } ], "md5sum": "a3a3de804f5a79ebc77a9c400350685a" }, { "dataPath": "params_shard_37.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.23.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.23.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.23.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.24.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14750720 }, { "name": "gpt_neox.layers.24.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14755840 }, { "name": "gpt_neox.layers.24.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.24.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.24.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 14771200 }, { "name": "gpt_neox.layers.24.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 24601600 }, { "name": "gpt_neox.layers.24.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 25830400 }, { "name": "gpt_neox.layers.24.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 25845760 }, { "name": "gpt_neox.layers.24.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 29122560 }, { "name": "gpt_neox.layers.24.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "058f1dcaa31f2ab220327255015f2412" }, { "dataPath": "params_shard_38.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.24.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.24.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.24.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.24.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.24.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.24.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29511680 }, { "name": "gpt_neox.layers.25.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29516800 }, { "name": "gpt_neox.layers.25.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29521920 }, { "name": "gpt_neox.layers.25.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29527040 }, { "name": "gpt_neox.layers.25.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "9d8251d5acaf40a6d75c63a527d28dd2" }, { "dataPath": "params_shard_39.bin", "format": "raw-shard", "nbytes": 29532160, "records": [ { "name": "gpt_neox.layers.25.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "gpt_neox.layers.25.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 9830400 }, { "name": "gpt_neox.layers.25.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 11059200 }, { "name": "gpt_neox.layers.25.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 11074560 }, { "name": "gpt_neox.layers.25.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 14351360 }, { "name": "gpt_neox.layers.25.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.25.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.25.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.25.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 29511680 } ], "md5sum": "2466e8a366df4a9178bf35f16a2f223b" }, { "dataPath": "params_shard_40.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.25.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.25.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.25.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.26.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14750720 }, { "name": "gpt_neox.layers.26.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14755840 }, { "name": "gpt_neox.layers.26.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.26.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.26.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 14771200 }, { "name": "gpt_neox.layers.26.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 24601600 }, { "name": "gpt_neox.layers.26.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 25830400 }, { "name": "gpt_neox.layers.26.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 25845760 }, { "name": "gpt_neox.layers.26.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 29122560 }, { "name": "gpt_neox.layers.26.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "c661caa057e341778c5474f48935c334" }, { "dataPath": "params_shard_41.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.26.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.26.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.26.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.26.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.26.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.26.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29511680 }, { "name": "gpt_neox.layers.27.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29516800 }, { "name": "gpt_neox.layers.27.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29521920 }, { "name": "gpt_neox.layers.27.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29527040 }, { "name": "gpt_neox.layers.27.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "b7f4080217d50d15a1fb177e376be422" }, { "dataPath": "params_shard_42.bin", "format": "raw-shard", "nbytes": 29532160, "records": [ { "name": "gpt_neox.layers.27.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "gpt_neox.layers.27.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 9830400 }, { "name": "gpt_neox.layers.27.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 11059200 }, { "name": "gpt_neox.layers.27.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 11074560 }, { "name": "gpt_neox.layers.27.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 14351360 }, { "name": "gpt_neox.layers.27.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.27.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.27.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.27.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 29511680 } ], "md5sum": "8516c97da19254fedaa584480e8182a3" }, { "dataPath": "params_shard_43.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.27.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.27.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.27.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.28.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14750720 }, { "name": "gpt_neox.layers.28.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14755840 }, { "name": "gpt_neox.layers.28.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.28.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.28.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 14771200 }, { "name": "gpt_neox.layers.28.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 24601600 }, { "name": "gpt_neox.layers.28.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 25830400 }, { "name": "gpt_neox.layers.28.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 25845760 }, { "name": "gpt_neox.layers.28.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 29122560 }, { "name": "gpt_neox.layers.28.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "d7446655f4dad3929d8482e8e8c0b0ab" }, { "dataPath": "params_shard_44.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.28.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.28.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.28.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.28.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.28.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.28.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29511680 }, { "name": "gpt_neox.layers.29.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29516800 }, { "name": "gpt_neox.layers.29.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29521920 }, { "name": "gpt_neox.layers.29.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29527040 }, { "name": "gpt_neox.layers.29.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "7a7b29d7074af62551b3dd4af0ad4f23" }, { "dataPath": "params_shard_45.bin", "format": "raw-shard", "nbytes": 29532160, "records": [ { "name": "gpt_neox.layers.29.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "gpt_neox.layers.29.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 9830400 }, { "name": "gpt_neox.layers.29.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 11059200 }, { "name": "gpt_neox.layers.29.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 11074560 }, { "name": "gpt_neox.layers.29.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 14351360 }, { "name": "gpt_neox.layers.29.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.29.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.29.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.29.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 29511680 } ], "md5sum": "39fab49b477c286079d3b16ca9bc0f93" }, { "dataPath": "params_shard_46.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.29.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.29.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.29.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.30.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14750720 }, { "name": "gpt_neox.layers.30.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14755840 }, { "name": "gpt_neox.layers.30.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.30.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.30.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 14771200 }, { "name": "gpt_neox.layers.30.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 24601600 }, { "name": "gpt_neox.layers.30.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 25830400 }, { "name": "gpt_neox.layers.30.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 25845760 }, { "name": "gpt_neox.layers.30.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 29122560 }, { "name": "gpt_neox.layers.30.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "199257245d44522f7b84668be1874ea4" }, { "dataPath": "params_shard_47.bin", "format": "raw-shard", "nbytes": 29537280, "records": [ { "name": "gpt_neox.layers.30.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.30.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.30.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 14745600 }, { "name": "gpt_neox.layers.30.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.30.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.30.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29511680 }, { "name": "gpt_neox.layers.31.input_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29516800 }, { "name": "gpt_neox.layers.31.input_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29521920 }, { "name": "gpt_neox.layers.31.post_attention_layernorm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29527040 }, { "name": "gpt_neox.layers.31.post_attention_layernorm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 29532160 } ], "md5sum": "fb185147691621efd7857e1594c41969" }, { "dataPath": "params_shard_48.bin", "format": "raw-shard", "nbytes": 29532160, "records": [ { "name": "gpt_neox.layers.31.attention.query_key_value.q_weight", "shape": [ 7680, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 9830400, "byteOffset": 0 }, { "name": "gpt_neox.layers.31.attention.query_key_value.q_scale", "shape": [ 7680, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1228800, "byteOffset": 9830400 }, { "name": "gpt_neox.layers.31.attention.query_key_value.bias", "shape": [ 7680 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 15360, "byteOffset": 11059200 }, { "name": "gpt_neox.layers.31.attention.dense.q_weight", "shape": [ 2560, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 3276800, "byteOffset": 11074560 }, { "name": "gpt_neox.layers.31.attention.dense.q_scale", "shape": [ 2560, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 409600, "byteOffset": 14351360 }, { "name": "gpt_neox.layers.31.attention.dense.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14760960 }, { "name": "gpt_neox.layers.31.mlp.dense_h_to_4h.q_weight", "shape": [ 10240, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 14766080 }, { "name": "gpt_neox.layers.31.mlp.dense_h_to_4h.q_scale", "shape": [ 10240, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 27873280 }, { "name": "gpt_neox.layers.31.mlp.dense_h_to_4h.bias", "shape": [ 10240 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 20480, "byteOffset": 29511680 } ], "md5sum": "544eb3e6c3be90f2b189cf94ee3b4996" }, { "dataPath": "params_shard_49.bin", "format": "raw-shard", "nbytes": 62914560, "records": [ { "name": "embed_out.q_weight", "shape": [ 49152, 320 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 62914560, "byteOffset": 0 } ], "md5sum": "4bbfa86554f4acd2d85bd88cfcf840c6" }, { "dataPath": "params_shard_50.bin", "format": "raw-shard", "nbytes": 22625280, "records": [ { "name": "gpt_neox.layers.31.mlp.dense_4h_to_h.q_weight", "shape": [ 2560, 1280 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 13107200, "byteOffset": 0 }, { "name": "gpt_neox.layers.31.mlp.dense_4h_to_h.q_scale", "shape": [ 2560, 320 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 1638400, "byteOffset": 13107200 }, { "name": "gpt_neox.layers.31.mlp.dense_4h_to_h.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14745600 }, { "name": "gpt_neox.final_layer_norm.weight", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14750720 }, { "name": "gpt_neox.final_layer_norm.bias", "shape": [ 2560 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 5120, "byteOffset": 14755840 }, { "name": "embed_out.q_scale", "shape": [ 49152, 80 ], "dtype": "float32", "format": "f32-to-bf16", "nbytes": 7864320, "byteOffset": 14760960 } ], "md5sum": "158c7cf77827663e898f016db7441589" } ] }