junrushao's picture
Initial commit
14f6523
raw
history blame contribute delete
204 kB
{
"metadata": {
"ParamSize": 518,
"ParamBytes": 1734430720.0,
"BitsPerParam": 5.010432978788311
},
"records": [
{
"dataPath": "params_shard_0.bin",
"format": "raw-shard",
"nbytes": 62914560,
"records": [
{
"name": "gpt_neox.embed_in.q_weight",
"shape": [
49152,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 62914560,
"byteOffset": 0
}
],
"md5sum": "130d0e0e3d1fa08301f4c0a224eb2a93"
},
{
"dataPath": "params_shard_1.bin",
"format": "raw-shard",
"nbytes": 22650880,
"records": [
{
"name": "gpt_neox.embed_in.q_scale",
"shape": [
49152,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 7864320,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.0.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 7864320
},
{
"name": "gpt_neox.layers.0.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 7869440
},
{
"name": "gpt_neox.layers.0.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 7874560
},
{
"name": "gpt_neox.layers.0.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 7879680
},
{
"name": "gpt_neox.layers.0.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 7884800
},
{
"name": "gpt_neox.layers.0.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 17715200
},
{
"name": "gpt_neox.layers.0.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 18944000
},
{
"name": "gpt_neox.layers.0.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 18959360
},
{
"name": "gpt_neox.layers.0.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 22236160
},
{
"name": "gpt_neox.layers.0.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 22645760
}
],
"md5sum": "18099736e1a0ff25881880e0accb4989"
},
{
"dataPath": "params_shard_2.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.0.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.0.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.0.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.0.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.0.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.0.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "gpt_neox.layers.1.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "gpt_neox.layers.1.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29521920
},
{
"name": "gpt_neox.layers.1.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29527040
},
{
"name": "gpt_neox.layers.1.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "23f6fc332f115869ac83adbd80a3f200"
},
{
"dataPath": "params_shard_3.bin",
"format": "raw-shard",
"nbytes": 29532160,
"records": [
{
"name": "gpt_neox.layers.1.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.1.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "gpt_neox.layers.1.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 11059200
},
{
"name": "gpt_neox.layers.1.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 11074560
},
{
"name": "gpt_neox.layers.1.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 14351360
},
{
"name": "gpt_neox.layers.1.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.1.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.1.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.1.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 29511680
}
],
"md5sum": "67d21fff48aa7f2a71f5412b16d16e1f"
},
{
"dataPath": "params_shard_4.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.1.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.1.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.1.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.2.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14750720
},
{
"name": "gpt_neox.layers.2.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14755840
},
{
"name": "gpt_neox.layers.2.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.2.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.2.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 14771200
},
{
"name": "gpt_neox.layers.2.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 24601600
},
{
"name": "gpt_neox.layers.2.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 25830400
},
{
"name": "gpt_neox.layers.2.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 25845760
},
{
"name": "gpt_neox.layers.2.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 29122560
},
{
"name": "gpt_neox.layers.2.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "83ccd56fcf9a6c83228308f5116c200f"
},
{
"dataPath": "params_shard_5.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.2.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.2.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.2.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.2.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.2.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.2.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "gpt_neox.layers.3.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "gpt_neox.layers.3.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29521920
},
{
"name": "gpt_neox.layers.3.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29527040
},
{
"name": "gpt_neox.layers.3.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "b024119fb95076ba55ab9be41f09ed96"
},
{
"dataPath": "params_shard_6.bin",
"format": "raw-shard",
"nbytes": 29532160,
"records": [
{
"name": "gpt_neox.layers.3.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.3.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "gpt_neox.layers.3.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 11059200
},
{
"name": "gpt_neox.layers.3.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 11074560
},
{
"name": "gpt_neox.layers.3.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 14351360
},
{
"name": "gpt_neox.layers.3.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.3.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.3.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.3.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 29511680
}
],
"md5sum": "d16bede9469ef0d657644d7b0b874953"
},
{
"dataPath": "params_shard_7.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.3.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.3.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.3.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.4.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14750720
},
{
"name": "gpt_neox.layers.4.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14755840
},
{
"name": "gpt_neox.layers.4.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.4.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.4.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 14771200
},
{
"name": "gpt_neox.layers.4.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 24601600
},
{
"name": "gpt_neox.layers.4.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 25830400
},
{
"name": "gpt_neox.layers.4.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 25845760
},
{
"name": "gpt_neox.layers.4.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 29122560
},
{
"name": "gpt_neox.layers.4.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "0273c2053de2bbba3c8addab8b51f29e"
},
{
"dataPath": "params_shard_8.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.4.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.4.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.4.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.4.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.4.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.4.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "gpt_neox.layers.5.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "gpt_neox.layers.5.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29521920
},
{
"name": "gpt_neox.layers.5.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29527040
},
{
"name": "gpt_neox.layers.5.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "c66d4d233e16e5564c8eb78db61a0ab0"
},
{
"dataPath": "params_shard_9.bin",
"format": "raw-shard",
"nbytes": 29532160,
"records": [
{
"name": "gpt_neox.layers.5.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.5.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "gpt_neox.layers.5.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 11059200
},
{
"name": "gpt_neox.layers.5.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 11074560
},
{
"name": "gpt_neox.layers.5.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 14351360
},
{
"name": "gpt_neox.layers.5.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.5.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.5.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.5.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 29511680
}
],
"md5sum": "6168569092ea967a6b0bc1d667eaabe5"
},
{
"dataPath": "params_shard_10.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.5.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.5.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.5.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.6.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14750720
},
{
"name": "gpt_neox.layers.6.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14755840
},
{
"name": "gpt_neox.layers.6.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.6.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.6.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 14771200
},
{
"name": "gpt_neox.layers.6.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 24601600
},
{
"name": "gpt_neox.layers.6.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 25830400
},
{
"name": "gpt_neox.layers.6.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 25845760
},
{
"name": "gpt_neox.layers.6.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 29122560
},
{
"name": "gpt_neox.layers.6.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "61d629151f9cf05ba8a6b9f156e91de5"
},
{
"dataPath": "params_shard_11.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.6.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.6.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.6.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.6.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.6.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.6.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "gpt_neox.layers.7.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "gpt_neox.layers.7.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29521920
},
{
"name": "gpt_neox.layers.7.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29527040
},
{
"name": "gpt_neox.layers.7.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "89e49c725e0aa441eef082b8aa282010"
},
{
"dataPath": "params_shard_12.bin",
"format": "raw-shard",
"nbytes": 29532160,
"records": [
{
"name": "gpt_neox.layers.7.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.7.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "gpt_neox.layers.7.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 11059200
},
{
"name": "gpt_neox.layers.7.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 11074560
},
{
"name": "gpt_neox.layers.7.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 14351360
},
{
"name": "gpt_neox.layers.7.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.7.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.7.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.7.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 29511680
}
],
"md5sum": "269f41cc257440a6bf82f5b2dc8b60e2"
},
{
"dataPath": "params_shard_13.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.7.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.7.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.7.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.8.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14750720
},
{
"name": "gpt_neox.layers.8.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14755840
},
{
"name": "gpt_neox.layers.8.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.8.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.8.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 14771200
},
{
"name": "gpt_neox.layers.8.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 24601600
},
{
"name": "gpt_neox.layers.8.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 25830400
},
{
"name": "gpt_neox.layers.8.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 25845760
},
{
"name": "gpt_neox.layers.8.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 29122560
},
{
"name": "gpt_neox.layers.8.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "bd24e6eb7b331cf9a28fdecadc179109"
},
{
"dataPath": "params_shard_14.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.8.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.8.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.8.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.8.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.8.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.8.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "gpt_neox.layers.9.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "gpt_neox.layers.9.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29521920
},
{
"name": "gpt_neox.layers.9.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29527040
},
{
"name": "gpt_neox.layers.9.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "6345876d27b4c166e3572ffe841159ad"
},
{
"dataPath": "params_shard_15.bin",
"format": "raw-shard",
"nbytes": 29532160,
"records": [
{
"name": "gpt_neox.layers.9.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.9.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "gpt_neox.layers.9.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 11059200
},
{
"name": "gpt_neox.layers.9.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 11074560
},
{
"name": "gpt_neox.layers.9.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 14351360
},
{
"name": "gpt_neox.layers.9.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.9.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.9.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.9.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 29511680
}
],
"md5sum": "19c19ef98553f64b91f1bed9b5ef4032"
},
{
"dataPath": "params_shard_16.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.9.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.9.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.9.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.10.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14750720
},
{
"name": "gpt_neox.layers.10.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14755840
},
{
"name": "gpt_neox.layers.10.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.10.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.10.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 14771200
},
{
"name": "gpt_neox.layers.10.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 24601600
},
{
"name": "gpt_neox.layers.10.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 25830400
},
{
"name": "gpt_neox.layers.10.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 25845760
},
{
"name": "gpt_neox.layers.10.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 29122560
},
{
"name": "gpt_neox.layers.10.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "c9e457cde4bd3fc22642f20cc4d19979"
},
{
"dataPath": "params_shard_17.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.10.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.10.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.10.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.10.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.10.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.10.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "gpt_neox.layers.11.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "gpt_neox.layers.11.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29521920
},
{
"name": "gpt_neox.layers.11.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29527040
},
{
"name": "gpt_neox.layers.11.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "af44068d1c581359eef7eb4f1080bb71"
},
{
"dataPath": "params_shard_18.bin",
"format": "raw-shard",
"nbytes": 29532160,
"records": [
{
"name": "gpt_neox.layers.11.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.11.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "gpt_neox.layers.11.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 11059200
},
{
"name": "gpt_neox.layers.11.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 11074560
},
{
"name": "gpt_neox.layers.11.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 14351360
},
{
"name": "gpt_neox.layers.11.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.11.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.11.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.11.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 29511680
}
],
"md5sum": "7f7ae45a99e46394d123cfe3c07af799"
},
{
"dataPath": "params_shard_19.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.11.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.11.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.11.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.12.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14750720
},
{
"name": "gpt_neox.layers.12.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14755840
},
{
"name": "gpt_neox.layers.12.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.12.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.12.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 14771200
},
{
"name": "gpt_neox.layers.12.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 24601600
},
{
"name": "gpt_neox.layers.12.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 25830400
},
{
"name": "gpt_neox.layers.12.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 25845760
},
{
"name": "gpt_neox.layers.12.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 29122560
},
{
"name": "gpt_neox.layers.12.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "17306d51500d13a9458adaa9a4d62826"
},
{
"dataPath": "params_shard_20.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.12.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.12.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.12.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.12.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.12.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.12.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "gpt_neox.layers.13.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "gpt_neox.layers.13.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29521920
},
{
"name": "gpt_neox.layers.13.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29527040
},
{
"name": "gpt_neox.layers.13.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "06b914336ed4f10684cbac356f7dfd5e"
},
{
"dataPath": "params_shard_21.bin",
"format": "raw-shard",
"nbytes": 29532160,
"records": [
{
"name": "gpt_neox.layers.13.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.13.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "gpt_neox.layers.13.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 11059200
},
{
"name": "gpt_neox.layers.13.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 11074560
},
{
"name": "gpt_neox.layers.13.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 14351360
},
{
"name": "gpt_neox.layers.13.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.13.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.13.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.13.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 29511680
}
],
"md5sum": "6e8ca0dd1339cfc6cf7d0bd7d15c37fb"
},
{
"dataPath": "params_shard_22.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.13.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.13.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.13.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.14.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14750720
},
{
"name": "gpt_neox.layers.14.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14755840
},
{
"name": "gpt_neox.layers.14.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.14.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.14.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 14771200
},
{
"name": "gpt_neox.layers.14.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 24601600
},
{
"name": "gpt_neox.layers.14.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 25830400
},
{
"name": "gpt_neox.layers.14.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 25845760
},
{
"name": "gpt_neox.layers.14.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 29122560
},
{
"name": "gpt_neox.layers.14.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "b53385ac21d45cbeaed1c341ac3e9eef"
},
{
"dataPath": "params_shard_23.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.14.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.14.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.14.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.14.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.14.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.14.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "gpt_neox.layers.15.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "gpt_neox.layers.15.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29521920
},
{
"name": "gpt_neox.layers.15.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29527040
},
{
"name": "gpt_neox.layers.15.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "64c15cf3bb3a4b225a26fe9d16e5b8e8"
},
{
"dataPath": "params_shard_24.bin",
"format": "raw-shard",
"nbytes": 29532160,
"records": [
{
"name": "gpt_neox.layers.15.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.15.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "gpt_neox.layers.15.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 11059200
},
{
"name": "gpt_neox.layers.15.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 11074560
},
{
"name": "gpt_neox.layers.15.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 14351360
},
{
"name": "gpt_neox.layers.15.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.15.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.15.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.15.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 29511680
}
],
"md5sum": "3fed9efcb61a25f0973ba392889dd85e"
},
{
"dataPath": "params_shard_25.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.15.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.15.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.15.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.16.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14750720
},
{
"name": "gpt_neox.layers.16.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14755840
},
{
"name": "gpt_neox.layers.16.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.16.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.16.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 14771200
},
{
"name": "gpt_neox.layers.16.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 24601600
},
{
"name": "gpt_neox.layers.16.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 25830400
},
{
"name": "gpt_neox.layers.16.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 25845760
},
{
"name": "gpt_neox.layers.16.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 29122560
},
{
"name": "gpt_neox.layers.16.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "7ccac867c6e382d0d542e8033ab3f775"
},
{
"dataPath": "params_shard_26.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.16.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.16.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.16.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.16.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.16.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.16.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "gpt_neox.layers.17.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "gpt_neox.layers.17.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29521920
},
{
"name": "gpt_neox.layers.17.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29527040
},
{
"name": "gpt_neox.layers.17.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "8bcd5c15ef2e875dfb94c4ebc1a209b3"
},
{
"dataPath": "params_shard_27.bin",
"format": "raw-shard",
"nbytes": 29532160,
"records": [
{
"name": "gpt_neox.layers.17.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.17.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "gpt_neox.layers.17.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 11059200
},
{
"name": "gpt_neox.layers.17.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 11074560
},
{
"name": "gpt_neox.layers.17.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 14351360
},
{
"name": "gpt_neox.layers.17.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.17.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.17.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.17.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 29511680
}
],
"md5sum": "214c1183504e11834ad5572e1a179d7a"
},
{
"dataPath": "params_shard_28.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.17.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.17.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.17.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.18.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14750720
},
{
"name": "gpt_neox.layers.18.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14755840
},
{
"name": "gpt_neox.layers.18.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.18.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.18.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 14771200
},
{
"name": "gpt_neox.layers.18.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 24601600
},
{
"name": "gpt_neox.layers.18.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 25830400
},
{
"name": "gpt_neox.layers.18.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 25845760
},
{
"name": "gpt_neox.layers.18.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 29122560
},
{
"name": "gpt_neox.layers.18.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "0f800a7329b9475d77a5d05f67cd5a5e"
},
{
"dataPath": "params_shard_29.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.18.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.18.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.18.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.18.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.18.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.18.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "gpt_neox.layers.19.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "gpt_neox.layers.19.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29521920
},
{
"name": "gpt_neox.layers.19.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29527040
},
{
"name": "gpt_neox.layers.19.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "0b34957f622e64e980722a305902a61a"
},
{
"dataPath": "params_shard_30.bin",
"format": "raw-shard",
"nbytes": 29532160,
"records": [
{
"name": "gpt_neox.layers.19.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.19.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "gpt_neox.layers.19.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 11059200
},
{
"name": "gpt_neox.layers.19.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 11074560
},
{
"name": "gpt_neox.layers.19.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 14351360
},
{
"name": "gpt_neox.layers.19.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.19.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.19.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.19.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 29511680
}
],
"md5sum": "0ba73f9c34dabeb77088a8b06f62635a"
},
{
"dataPath": "params_shard_31.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.19.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.19.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.19.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.20.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14750720
},
{
"name": "gpt_neox.layers.20.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14755840
},
{
"name": "gpt_neox.layers.20.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.20.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.20.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 14771200
},
{
"name": "gpt_neox.layers.20.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 24601600
},
{
"name": "gpt_neox.layers.20.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 25830400
},
{
"name": "gpt_neox.layers.20.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 25845760
},
{
"name": "gpt_neox.layers.20.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 29122560
},
{
"name": "gpt_neox.layers.20.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "77b6f858718a34d99d0d583dd2c3d2f6"
},
{
"dataPath": "params_shard_32.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.20.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.20.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.20.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.20.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.20.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.20.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "gpt_neox.layers.21.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "gpt_neox.layers.21.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29521920
},
{
"name": "gpt_neox.layers.21.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29527040
},
{
"name": "gpt_neox.layers.21.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "09216ea1e891008fc396c956c828131b"
},
{
"dataPath": "params_shard_33.bin",
"format": "raw-shard",
"nbytes": 29532160,
"records": [
{
"name": "gpt_neox.layers.21.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.21.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "gpt_neox.layers.21.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 11059200
},
{
"name": "gpt_neox.layers.21.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 11074560
},
{
"name": "gpt_neox.layers.21.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 14351360
},
{
"name": "gpt_neox.layers.21.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.21.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.21.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.21.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 29511680
}
],
"md5sum": "91242d7567d250a96dd0e05c62ec787e"
},
{
"dataPath": "params_shard_34.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.21.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.21.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.21.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.22.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14750720
},
{
"name": "gpt_neox.layers.22.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14755840
},
{
"name": "gpt_neox.layers.22.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.22.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.22.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 14771200
},
{
"name": "gpt_neox.layers.22.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 24601600
},
{
"name": "gpt_neox.layers.22.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 25830400
},
{
"name": "gpt_neox.layers.22.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 25845760
},
{
"name": "gpt_neox.layers.22.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 29122560
},
{
"name": "gpt_neox.layers.22.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "61f7cae329b8d1bdfed25bb7f13d262e"
},
{
"dataPath": "params_shard_35.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.22.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.22.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.22.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.22.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.22.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.22.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "gpt_neox.layers.23.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "gpt_neox.layers.23.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29521920
},
{
"name": "gpt_neox.layers.23.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29527040
},
{
"name": "gpt_neox.layers.23.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "3308bb6141a885c8cdd40d80092993ef"
},
{
"dataPath": "params_shard_36.bin",
"format": "raw-shard",
"nbytes": 29532160,
"records": [
{
"name": "gpt_neox.layers.23.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.23.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "gpt_neox.layers.23.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 11059200
},
{
"name": "gpt_neox.layers.23.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 11074560
},
{
"name": "gpt_neox.layers.23.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 14351360
},
{
"name": "gpt_neox.layers.23.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.23.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.23.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.23.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 29511680
}
],
"md5sum": "a3a3de804f5a79ebc77a9c400350685a"
},
{
"dataPath": "params_shard_37.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.23.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.23.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.23.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.24.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14750720
},
{
"name": "gpt_neox.layers.24.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14755840
},
{
"name": "gpt_neox.layers.24.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.24.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.24.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 14771200
},
{
"name": "gpt_neox.layers.24.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 24601600
},
{
"name": "gpt_neox.layers.24.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 25830400
},
{
"name": "gpt_neox.layers.24.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 25845760
},
{
"name": "gpt_neox.layers.24.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 29122560
},
{
"name": "gpt_neox.layers.24.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "058f1dcaa31f2ab220327255015f2412"
},
{
"dataPath": "params_shard_38.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.24.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.24.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.24.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.24.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.24.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.24.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "gpt_neox.layers.25.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "gpt_neox.layers.25.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29521920
},
{
"name": "gpt_neox.layers.25.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29527040
},
{
"name": "gpt_neox.layers.25.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "9d8251d5acaf40a6d75c63a527d28dd2"
},
{
"dataPath": "params_shard_39.bin",
"format": "raw-shard",
"nbytes": 29532160,
"records": [
{
"name": "gpt_neox.layers.25.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.25.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "gpt_neox.layers.25.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 11059200
},
{
"name": "gpt_neox.layers.25.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 11074560
},
{
"name": "gpt_neox.layers.25.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 14351360
},
{
"name": "gpt_neox.layers.25.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.25.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.25.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.25.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 29511680
}
],
"md5sum": "2466e8a366df4a9178bf35f16a2f223b"
},
{
"dataPath": "params_shard_40.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.25.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.25.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.25.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.26.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14750720
},
{
"name": "gpt_neox.layers.26.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14755840
},
{
"name": "gpt_neox.layers.26.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.26.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.26.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 14771200
},
{
"name": "gpt_neox.layers.26.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 24601600
},
{
"name": "gpt_neox.layers.26.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 25830400
},
{
"name": "gpt_neox.layers.26.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 25845760
},
{
"name": "gpt_neox.layers.26.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 29122560
},
{
"name": "gpt_neox.layers.26.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "c661caa057e341778c5474f48935c334"
},
{
"dataPath": "params_shard_41.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.26.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.26.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.26.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.26.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.26.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.26.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "gpt_neox.layers.27.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "gpt_neox.layers.27.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29521920
},
{
"name": "gpt_neox.layers.27.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29527040
},
{
"name": "gpt_neox.layers.27.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "b7f4080217d50d15a1fb177e376be422"
},
{
"dataPath": "params_shard_42.bin",
"format": "raw-shard",
"nbytes": 29532160,
"records": [
{
"name": "gpt_neox.layers.27.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.27.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "gpt_neox.layers.27.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 11059200
},
{
"name": "gpt_neox.layers.27.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 11074560
},
{
"name": "gpt_neox.layers.27.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 14351360
},
{
"name": "gpt_neox.layers.27.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.27.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.27.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.27.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 29511680
}
],
"md5sum": "8516c97da19254fedaa584480e8182a3"
},
{
"dataPath": "params_shard_43.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.27.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.27.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.27.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.28.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14750720
},
{
"name": "gpt_neox.layers.28.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14755840
},
{
"name": "gpt_neox.layers.28.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.28.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.28.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 14771200
},
{
"name": "gpt_neox.layers.28.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 24601600
},
{
"name": "gpt_neox.layers.28.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 25830400
},
{
"name": "gpt_neox.layers.28.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 25845760
},
{
"name": "gpt_neox.layers.28.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 29122560
},
{
"name": "gpt_neox.layers.28.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "d7446655f4dad3929d8482e8e8c0b0ab"
},
{
"dataPath": "params_shard_44.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.28.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.28.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.28.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.28.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.28.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.28.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "gpt_neox.layers.29.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "gpt_neox.layers.29.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29521920
},
{
"name": "gpt_neox.layers.29.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29527040
},
{
"name": "gpt_neox.layers.29.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "7a7b29d7074af62551b3dd4af0ad4f23"
},
{
"dataPath": "params_shard_45.bin",
"format": "raw-shard",
"nbytes": 29532160,
"records": [
{
"name": "gpt_neox.layers.29.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.29.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "gpt_neox.layers.29.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 11059200
},
{
"name": "gpt_neox.layers.29.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 11074560
},
{
"name": "gpt_neox.layers.29.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 14351360
},
{
"name": "gpt_neox.layers.29.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.29.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.29.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.29.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 29511680
}
],
"md5sum": "39fab49b477c286079d3b16ca9bc0f93"
},
{
"dataPath": "params_shard_46.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.29.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.29.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.29.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.30.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14750720
},
{
"name": "gpt_neox.layers.30.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14755840
},
{
"name": "gpt_neox.layers.30.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.30.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.30.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 14771200
},
{
"name": "gpt_neox.layers.30.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 24601600
},
{
"name": "gpt_neox.layers.30.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 25830400
},
{
"name": "gpt_neox.layers.30.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 25845760
},
{
"name": "gpt_neox.layers.30.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 29122560
},
{
"name": "gpt_neox.layers.30.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "199257245d44522f7b84668be1874ea4"
},
{
"dataPath": "params_shard_47.bin",
"format": "raw-shard",
"nbytes": 29537280,
"records": [
{
"name": "gpt_neox.layers.30.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.30.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.30.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 14745600
},
{
"name": "gpt_neox.layers.30.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.30.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.30.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29511680
},
{
"name": "gpt_neox.layers.31.input_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29516800
},
{
"name": "gpt_neox.layers.31.input_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29521920
},
{
"name": "gpt_neox.layers.31.post_attention_layernorm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29527040
},
{
"name": "gpt_neox.layers.31.post_attention_layernorm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 29532160
}
],
"md5sum": "fb185147691621efd7857e1594c41969"
},
{
"dataPath": "params_shard_48.bin",
"format": "raw-shard",
"nbytes": 29532160,
"records": [
{
"name": "gpt_neox.layers.31.attention.query_key_value.q_weight",
"shape": [
7680,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 9830400,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.31.attention.query_key_value.q_scale",
"shape": [
7680,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1228800,
"byteOffset": 9830400
},
{
"name": "gpt_neox.layers.31.attention.query_key_value.bias",
"shape": [
7680
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 15360,
"byteOffset": 11059200
},
{
"name": "gpt_neox.layers.31.attention.dense.q_weight",
"shape": [
2560,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 3276800,
"byteOffset": 11074560
},
{
"name": "gpt_neox.layers.31.attention.dense.q_scale",
"shape": [
2560,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 409600,
"byteOffset": 14351360
},
{
"name": "gpt_neox.layers.31.attention.dense.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14760960
},
{
"name": "gpt_neox.layers.31.mlp.dense_h_to_4h.q_weight",
"shape": [
10240,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 14766080
},
{
"name": "gpt_neox.layers.31.mlp.dense_h_to_4h.q_scale",
"shape": [
10240,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 27873280
},
{
"name": "gpt_neox.layers.31.mlp.dense_h_to_4h.bias",
"shape": [
10240
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 20480,
"byteOffset": 29511680
}
],
"md5sum": "544eb3e6c3be90f2b189cf94ee3b4996"
},
{
"dataPath": "params_shard_49.bin",
"format": "raw-shard",
"nbytes": 62914560,
"records": [
{
"name": "embed_out.q_weight",
"shape": [
49152,
320
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 62914560,
"byteOffset": 0
}
],
"md5sum": "4bbfa86554f4acd2d85bd88cfcf840c6"
},
{
"dataPath": "params_shard_50.bin",
"format": "raw-shard",
"nbytes": 22625280,
"records": [
{
"name": "gpt_neox.layers.31.mlp.dense_4h_to_h.q_weight",
"shape": [
2560,
1280
],
"dtype": "uint32",
"format": "f32-to-bf16",
"nbytes": 13107200,
"byteOffset": 0
},
{
"name": "gpt_neox.layers.31.mlp.dense_4h_to_h.q_scale",
"shape": [
2560,
320
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 1638400,
"byteOffset": 13107200
},
{
"name": "gpt_neox.layers.31.mlp.dense_4h_to_h.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14745600
},
{
"name": "gpt_neox.final_layer_norm.weight",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14750720
},
{
"name": "gpt_neox.final_layer_norm.bias",
"shape": [
2560
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 5120,
"byteOffset": 14755840
},
{
"name": "embed_out.q_scale",
"shape": [
49152,
80
],
"dtype": "bfloat16",
"format": "raw",
"nbytes": 7864320,
"byteOffset": 14760960
}
],
"md5sum": "158c7cf77827663e898f016db7441589"
}
]
}