{ "metadata": { "ParamSize": 313, "ParamBytes": 4284263424.0, "BitsPerParam": 4.500503319461262 }, "records": [ { "dataPath": "params_shard_0.bin", "format": "raw-shard", "nbytes": 272498688, "records": [ { "name": "lm_head.q_weight", "shape": [ 152064, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 272498688, "byteOffset": 0 } ], "md5sum": "ff93a9f5978a7f750c8ccb437051d4b5" }, { "dataPath": "params_shard_1.bin", "format": "raw-shard", "nbytes": 34062336, "records": [ { "name": "lm_head.q_scale", "shape": [ 152064, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 34062336, "byteOffset": 0 } ], "md5sum": "fb2bb94fc27d23088991dd26dc5a8bda" }, { "dataPath": "params_shard_2.bin", "format": "raw-shard", "nbytes": 272498688, "records": [ { "name": "model.embed_tokens.q_weight", "shape": [ 152064, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 272498688, "byteOffset": 0 } ], "md5sum": "6aa3867b6f383f02d4286173f816f0eb" }, { "dataPath": "params_shard_3.bin", "format": "raw-shard", "nbytes": 34062336, "records": [ { "name": "model.embed_tokens.q_scale", "shape": [ 152064, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 34062336, "byteOffset": 0 } ], "md5sum": "3cbe26a08e2d7a4534b916da4be16b79" }, { "dataPath": "params_shard_4.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.0.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "110310c23b819cf12c40716a9495a9c0" }, { "dataPath": "params_shard_5.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.0.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "ee582b01ef03c67fffb04ca3e106821b" }, { "dataPath": "params_shard_6.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.1.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "26461dc39a102f6497fb5d9dd31e01c3" }, { "dataPath": "params_shard_7.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.1.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "bed37f352041fa049c339cc56b6565f9" }, { "dataPath": "params_shard_8.bin", "format": "raw-shard", "nbytes": 33519616, "records": [ { "name": "model.layers.0.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 0 }, { "name": "model.layers.0.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 7168 }, { "name": "model.layers.0.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 4250624 }, { "name": "model.layers.0.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 12737536 }, { "name": "model.layers.0.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 12744704 }, { "name": "model.layers.0.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 12753920 }, { "name": "model.layers.0.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 21011456 }, { "name": "model.layers.0.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 22043648 }, { "name": "model.layers.0.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 28466176 }, { "name": "model.layers.1.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 29268992 }, { "name": "model.layers.1.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 29276160 } ], "md5sum": "1de4b095e1ac7c44afba2d1a59be89d0" }, { "dataPath": "params_shard_9.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.2.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "c15b3e170a9749aff3759970bd63a6c4" }, { "dataPath": "params_shard_10.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.2.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "287b2288386093ef0ccbe4f9ce8ea0be" }, { "dataPath": "params_shard_11.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.1.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 0 }, { "name": "model.layers.1.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 8486912 }, { "name": "model.layers.1.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 8494080 }, { "name": "model.layers.1.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 8503296 }, { "name": "model.layers.1.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 16760832 }, { "name": "model.layers.1.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 17793024 }, { "name": "model.layers.1.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 24215552 }, { "name": "model.layers.2.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 25018368 }, { "name": "model.layers.2.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 25025536 } ], "md5sum": "031fb12a46593dbe99246a681c7a19ac" }, { "dataPath": "params_shard_12.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.3.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "33bf6b6d7fae7aaa42222386075c3ed2" }, { "dataPath": "params_shard_13.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.3.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "6e55755c50ad6743450c2dd459364e62" }, { "dataPath": "params_shard_14.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.2.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 0 }, { "name": "model.layers.2.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 8486912 }, { "name": "model.layers.2.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 8494080 }, { "name": "model.layers.2.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 8503296 }, { "name": "model.layers.2.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 16760832 }, { "name": "model.layers.2.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 17793024 }, { "name": "model.layers.2.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 24215552 }, { "name": "model.layers.3.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 25018368 }, { "name": "model.layers.3.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 25025536 } ], "md5sum": "9be40d6a0f57b4272c0e73992020cfba" }, { "dataPath": "params_shard_15.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.4.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "e254dce7fd94fc9fb2689f46d4f267cb" }, { "dataPath": "params_shard_16.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.4.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "3c1ac4361ae532d437de2e338f3d1950" }, { "dataPath": "params_shard_17.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.3.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 0 }, { "name": "model.layers.3.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 8486912 }, { "name": "model.layers.3.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 8494080 }, { "name": "model.layers.3.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 8503296 }, { "name": "model.layers.3.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 16760832 }, { "name": "model.layers.3.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 17793024 }, { "name": "model.layers.3.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 24215552 }, { "name": "model.layers.4.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 25018368 }, { "name": "model.layers.4.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 25025536 } ], "md5sum": "07871b460d9d55468dbb986dbbe45556" }, { "dataPath": "params_shard_18.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.5.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "7edcc35606ab0457e308f51811de224c" }, { "dataPath": "params_shard_19.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.5.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "9109b0a2c4cf2e2b448b681437e0a75d" }, { "dataPath": "params_shard_20.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.4.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 0 }, { "name": "model.layers.4.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 8486912 }, { "name": "model.layers.4.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 8494080 }, { "name": "model.layers.4.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 8503296 }, { "name": "model.layers.4.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 16760832 }, { "name": "model.layers.4.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 17793024 }, { "name": "model.layers.4.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 24215552 }, { "name": "model.layers.5.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 25018368 }, { "name": "model.layers.5.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 25025536 } ], "md5sum": "dbebd9b3579154ed3423af4ad1b330fe" }, { "dataPath": "params_shard_21.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.6.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "f311fa34349eeff5077441222792049a" }, { "dataPath": "params_shard_22.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.6.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "5def2f0e12f847435fde98022e215ede" }, { "dataPath": "params_shard_23.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.5.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 0 }, { "name": "model.layers.5.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 8486912 }, { "name": "model.layers.5.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 8494080 }, { "name": "model.layers.5.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 8503296 }, { "name": "model.layers.5.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 16760832 }, { "name": "model.layers.5.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 17793024 }, { "name": "model.layers.5.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 24215552 }, { "name": "model.layers.6.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 25018368 }, { "name": "model.layers.6.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 25025536 } ], "md5sum": "c643985f395be7acfd78b29715e27483" }, { "dataPath": "params_shard_24.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.7.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "7d14f56c49053a9058f8073a2c7b1373" }, { "dataPath": "params_shard_25.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.7.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "22bbaac474d9b6bfc1d62cd743c43f3b" }, { "dataPath": "params_shard_26.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.6.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 0 }, { "name": "model.layers.6.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 8486912 }, { "name": "model.layers.6.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 8494080 }, { "name": "model.layers.6.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 8503296 }, { "name": "model.layers.6.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 16760832 }, { "name": "model.layers.6.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 17793024 }, { "name": "model.layers.6.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 24215552 }, { "name": "model.layers.7.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 25018368 }, { "name": "model.layers.7.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 25025536 } ], "md5sum": "319d31e731aa7ebd95eb4886ba082b4e" }, { "dataPath": "params_shard_27.bin", "format": "raw-shard", "nbytes": 33285120, "records": [ { "name": "model.layers.7.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 0 }, { "name": "model.layers.7.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 8486912 }, { "name": "model.layers.7.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 8494080 }, { "name": "model.layers.7.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 8503296 }, { "name": "model.layers.7.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 16760832 }, { "name": "model.layers.7.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 17793024 }, { "name": "model.layers.7.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 24215552 }, { "name": "model.layers.8.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 25018368 }, { "name": "model.layers.8.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 25027584 } ], "md5sum": "7f9a5d0e40f4d04b945a3b1e02542799" }, { "dataPath": "params_shard_28.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.10.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "9719343c858deaa671ac7a57a8bbb68d" }, { "dataPath": "params_shard_29.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.10.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "dfef6cf46b84875101c4dfb8e50e3ac1" }, { "dataPath": "params_shard_30.bin", "format": "raw-shard", "nbytes": 30301184, "records": [ { "name": "model.layers.8.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 0 }, { "name": "model.layers.8.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 1032192 }, { "name": "model.layers.8.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 7454720 }, { "name": "model.layers.10.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 8257536 }, { "name": "model.layers.10.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 8264704 }, { "name": "model.layers.10.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 12508160 }, { "name": "model.layers.10.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 20995072 }, { "name": "model.layers.10.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 21002240 }, { "name": "model.layers.10.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 21011456 }, { "name": "model.layers.10.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 29268992 } ], "md5sum": "ed72f389ff84c5e305d2c040470786d7" }, { "dataPath": "params_shard_31.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.11.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "4ea35f5ebf4bd73277b9396faf1d1d34" }, { "dataPath": "params_shard_32.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.11.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "9b6fbb2b44d8097986641def0eeecea9" }, { "dataPath": "params_shard_33.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.10.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 0 }, { "name": "model.layers.10.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 6422528 }, { "name": "model.layers.11.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 7225344 }, { "name": "model.layers.11.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 7232512 }, { "name": "model.layers.11.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 11475968 }, { "name": "model.layers.11.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 19962880 }, { "name": "model.layers.11.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 19970048 }, { "name": "model.layers.11.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 19979264 }, { "name": "model.layers.11.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 28236800 } ], "md5sum": "5d9e10c7f0df90c10a596194fb16b7fa" }, { "dataPath": "params_shard_34.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.12.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "eb8f7a1ce79922b81aca3301d19bcadf" }, { "dataPath": "params_shard_35.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.12.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "ed5980d4d89cc8acf9823d0317828021" }, { "dataPath": "params_shard_36.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.11.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 0 }, { "name": "model.layers.11.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 6422528 }, { "name": "model.layers.12.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 7225344 }, { "name": "model.layers.12.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 7232512 }, { "name": "model.layers.12.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 11475968 }, { "name": "model.layers.12.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 19962880 }, { "name": "model.layers.12.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 19970048 }, { "name": "model.layers.12.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 19979264 }, { "name": "model.layers.12.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 28236800 } ], "md5sum": "d218a339c57ef59a59fc4ff002fcb901" }, { "dataPath": "params_shard_37.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.13.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "36c499fa48847aabdaceab0a3e766c6a" }, { "dataPath": "params_shard_38.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.13.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "7e175d4bcfeb2d445d5bf97f2483ff38" }, { "dataPath": "params_shard_39.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.12.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 0 }, { "name": "model.layers.12.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 6422528 }, { "name": "model.layers.13.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 7225344 }, { "name": "model.layers.13.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 7232512 }, { "name": "model.layers.13.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 11475968 }, { "name": "model.layers.13.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 19962880 }, { "name": "model.layers.13.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 19970048 }, { "name": "model.layers.13.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 19979264 }, { "name": "model.layers.13.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 28236800 } ], "md5sum": "5c54e28efb79e0023f20312e56506ce2" }, { "dataPath": "params_shard_40.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.14.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "d6a9427b9f3cdbfd1c7b61677c23cc77" }, { "dataPath": "params_shard_41.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.14.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "eec29b7444d6f1f30efc17c9d47c6529" }, { "dataPath": "params_shard_42.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.13.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 0 }, { "name": "model.layers.13.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 6422528 }, { "name": "model.layers.14.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 7225344 }, { "name": "model.layers.14.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 7232512 }, { "name": "model.layers.14.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 11475968 }, { "name": "model.layers.14.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 19962880 }, { "name": "model.layers.14.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 19970048 }, { "name": "model.layers.14.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 19979264 }, { "name": "model.layers.14.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 28236800 } ], "md5sum": "d935b382db625327d974dfc6f4169cc7" }, { "dataPath": "params_shard_43.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.15.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "68fd4d0d85be11bd52b59265aa368151" }, { "dataPath": "params_shard_44.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.15.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "01740237bce31becf0b19068a4084861" }, { "dataPath": "params_shard_45.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.14.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 0 }, { "name": "model.layers.14.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 6422528 }, { "name": "model.layers.15.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 7225344 }, { "name": "model.layers.15.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 7232512 }, { "name": "model.layers.15.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 11475968 }, { "name": "model.layers.15.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 19962880 }, { "name": "model.layers.15.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 19970048 }, { "name": "model.layers.15.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 19979264 }, { "name": "model.layers.15.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 28236800 } ], "md5sum": "e2feba0cc14fb46cdd4dc28e24d32111" }, { "dataPath": "params_shard_46.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.16.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "d002b3695edc1f8215d3301ecb76e595" }, { "dataPath": "params_shard_47.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.16.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "4c5da4ecd800c6cfb774387dbc00e6ad" }, { "dataPath": "params_shard_48.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.15.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 0 }, { "name": "model.layers.15.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 6422528 }, { "name": "model.layers.16.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 7225344 }, { "name": "model.layers.16.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 7232512 }, { "name": "model.layers.16.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 11475968 }, { "name": "model.layers.16.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 19962880 }, { "name": "model.layers.16.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 19970048 }, { "name": "model.layers.16.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 19979264 }, { "name": "model.layers.16.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 28236800 } ], "md5sum": "da13ad0ddcf68cbf3a8fe1f98b078936" }, { "dataPath": "params_shard_49.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.17.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "cd69b17034dfa01c5dddf5062802e14b" }, { "dataPath": "params_shard_50.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.17.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "6ebd919c6d56b10a9c652714327f8d70" }, { "dataPath": "params_shard_51.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.16.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 0 }, { "name": "model.layers.16.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 6422528 }, { "name": "model.layers.17.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 7225344 }, { "name": "model.layers.17.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 7232512 }, { "name": "model.layers.17.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 11475968 }, { "name": "model.layers.17.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 19962880 }, { "name": "model.layers.17.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 19970048 }, { "name": "model.layers.17.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 19979264 }, { "name": "model.layers.17.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 28236800 } ], "md5sum": "a9b06531cb473e8b7891c55b369c6e12" }, { "dataPath": "params_shard_52.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.18.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "cc8fb5b9a98ca74bf80dea6812d61b01" }, { "dataPath": "params_shard_53.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.8.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "6d1a84a2e02ef6a2e7be0e753e1d30aa" }, { "dataPath": "params_shard_54.bin", "format": "raw-shard", "nbytes": 32243712, "records": [ { "name": "model.layers.17.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 0 }, { "name": "model.layers.17.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 6422528 }, { "name": "model.layers.18.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 7225344 }, { "name": "model.layers.18.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 15712256 }, { "name": "model.layers.18.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 15721472 }, { "name": "model.layers.18.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 23979008 }, { "name": "model.layers.18.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 25011200 }, { "name": "model.layers.18.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 31433728 }, { "name": "model.layers.8.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 32236544 } ], "md5sum": "1889a6556905f953f93f0050e0200583" }, { "dataPath": "params_shard_55.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.8.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "a3e8eecc9792ce5a615665ac4d78cfdb" }, { "dataPath": "params_shard_56.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.9.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "0a33d096b4f51ea3dddc905fd297d34a" }, { "dataPath": "params_shard_57.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.9.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "070fc6ba14e6ff2a83cd164879a51f13" }, { "dataPath": "params_shard_58.bin", "format": "raw-shard", "nbytes": 25491456, "records": [ { "name": "model.layers.8.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 0 }, { "name": "model.layers.8.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 4243456 }, { "name": "model.layers.8.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 12730368 }, { "name": "model.layers.9.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 12737536 }, { "name": "model.layers.9.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 12744704 }, { "name": "model.layers.9.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 16988160 }, { "name": "model.layers.9.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 25475072 }, { "name": "model.layers.9.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 25482240 } ], "md5sum": "cea794aaddac81efae198df16597bd38" }, { "dataPath": "params_shard_59.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.18.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "c6173e184bc85c9fda351292a6dc25ed" }, { "dataPath": "params_shard_60.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.19.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "25de9b12c0a4c2a4bc7acce587b7005a" }, { "dataPath": "params_shard_61.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.19.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "510259ffa4e825af8ed39501fa3871fc" }, { "dataPath": "params_shard_62.bin", "format": "raw-shard", "nbytes": 33526784, "records": [ { "name": "model.layers.9.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 0 }, { "name": "model.layers.9.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 8257536 }, { "name": "model.layers.9.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 9289728 }, { "name": "model.layers.9.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 15712256 }, { "name": "model.layers.18.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 16515072 }, { "name": "model.layers.18.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 16522240 }, { "name": "model.layers.18.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 20765696 }, { "name": "model.layers.19.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 20772864 }, { "name": "model.layers.19.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 20780032 }, { "name": "model.layers.19.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 25023488 }, { "name": "model.layers.19.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 33510400 }, { "name": "model.layers.19.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 33517568 } ], "md5sum": "92381a94c1f2a011765c8430b5275939" }, { "dataPath": "params_shard_63.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.20.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "27c524e3eb6de2170bd0878bbc5a05ae" }, { "dataPath": "params_shard_64.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.20.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "8fdec9392302c7c637e9e0260a6c8b42" }, { "dataPath": "params_shard_65.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.19.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 0 }, { "name": "model.layers.19.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 8257536 }, { "name": "model.layers.19.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 9289728 }, { "name": "model.layers.19.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 15712256 }, { "name": "model.layers.20.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 16515072 }, { "name": "model.layers.20.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 16522240 }, { "name": "model.layers.20.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 20765696 }, { "name": "model.layers.20.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 29252608 }, { "name": "model.layers.20.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 29259776 } ], "md5sum": "2363bd1ce1528afca752f27686b90891" }, { "dataPath": "params_shard_66.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.21.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "0d5064048059c1e1f47306c0a0f495a7" }, { "dataPath": "params_shard_67.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.21.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "4316acaaab483a869e1ffbfeece538c2" }, { "dataPath": "params_shard_68.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.20.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 0 }, { "name": "model.layers.20.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 8257536 }, { "name": "model.layers.20.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 9289728 }, { "name": "model.layers.20.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 15712256 }, { "name": "model.layers.21.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 16515072 }, { "name": "model.layers.21.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 16522240 }, { "name": "model.layers.21.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 20765696 }, { "name": "model.layers.21.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 29252608 }, { "name": "model.layers.21.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 29259776 } ], "md5sum": "f782ac3725ff0cbb210486c97fde29f3" }, { "dataPath": "params_shard_69.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.22.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "96d53f37565f514fce1cf49f530f9192" }, { "dataPath": "params_shard_70.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.22.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "862728476ed639dc49fd6aa17daf3b8c" }, { "dataPath": "params_shard_71.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.21.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 0 }, { "name": "model.layers.21.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 8257536 }, { "name": "model.layers.21.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 9289728 }, { "name": "model.layers.21.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 15712256 }, { "name": "model.layers.22.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 16515072 }, { "name": "model.layers.22.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 16522240 }, { "name": "model.layers.22.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 20765696 }, { "name": "model.layers.22.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 29252608 }, { "name": "model.layers.22.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 29259776 } ], "md5sum": "24859bc227bf9649f7c66822f94e0dd1" }, { "dataPath": "params_shard_72.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.23.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "68c36634f1bb287048c40dc4bdbffd43" }, { "dataPath": "params_shard_73.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.23.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "12f32e7464a1a4fb19a0a897914ea430" }, { "dataPath": "params_shard_74.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.22.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 0 }, { "name": "model.layers.22.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 8257536 }, { "name": "model.layers.22.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 9289728 }, { "name": "model.layers.22.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 15712256 }, { "name": "model.layers.23.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 16515072 }, { "name": "model.layers.23.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 16522240 }, { "name": "model.layers.23.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 20765696 }, { "name": "model.layers.23.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 29252608 }, { "name": "model.layers.23.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 29259776 } ], "md5sum": "828df638df4bbeb4a6dd614fb77ff128" }, { "dataPath": "params_shard_75.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.24.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "30cbcaf64fc43c540a16032aebe3fb9d" }, { "dataPath": "params_shard_76.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.24.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "d75891ab670f20779f25087c90a8864f" }, { "dataPath": "params_shard_77.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.23.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 0 }, { "name": "model.layers.23.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 8257536 }, { "name": "model.layers.23.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 9289728 }, { "name": "model.layers.23.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 15712256 }, { "name": "model.layers.24.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 16515072 }, { "name": "model.layers.24.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 16522240 }, { "name": "model.layers.24.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 20765696 }, { "name": "model.layers.24.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 29252608 }, { "name": "model.layers.24.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 29259776 } ], "md5sum": "ed8e0f4a8e1f633d1fe4ade55f48390d" }, { "dataPath": "params_shard_78.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.25.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "a06b3594e46cba82a0fe423884e44482" }, { "dataPath": "params_shard_79.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.25.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "e27bbb9b1c770f7951e891f36668a280" }, { "dataPath": "params_shard_80.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.24.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 0 }, { "name": "model.layers.24.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 8257536 }, { "name": "model.layers.24.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 9289728 }, { "name": "model.layers.24.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 15712256 }, { "name": "model.layers.25.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 16515072 }, { "name": "model.layers.25.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 16522240 }, { "name": "model.layers.25.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 20765696 }, { "name": "model.layers.25.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 29252608 }, { "name": "model.layers.25.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 29259776 } ], "md5sum": "fa9814fd3ec001375492952d6ab234e5" }, { "dataPath": "params_shard_81.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.26.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "f9c58eb62c14343bcb0f1ce3bb0902d9" }, { "dataPath": "params_shard_82.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.26.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "98d47033b29d7c1b963e03079ae3e271" }, { "dataPath": "params_shard_83.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.25.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 0 }, { "name": "model.layers.25.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 8257536 }, { "name": "model.layers.25.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 9289728 }, { "name": "model.layers.25.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 15712256 }, { "name": "model.layers.26.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 16515072 }, { "name": "model.layers.26.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 16522240 }, { "name": "model.layers.26.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 20765696 }, { "name": "model.layers.26.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 29252608 }, { "name": "model.layers.26.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 29259776 } ], "md5sum": "58479984157ade7c733609410f864830" }, { "dataPath": "params_shard_84.bin", "format": "raw-shard", "nbytes": 33947648, "records": [ { "name": "model.layers.27.mlp.down_proj.q_weight", "shape": [ 3584, 2368 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 33947648, "byteOffset": 0 } ], "md5sum": "f4a0755a441820cc7a2c5703f56afc09" }, { "dataPath": "params_shard_85.bin", "format": "raw-shard", "nbytes": 67895296, "records": [ { "name": "model.layers.27.mlp.gate_up_proj.q_weight", "shape": [ 37888, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 67895296, "byteOffset": 0 } ], "md5sum": "6a35bc13fb662f480e5bfabbc15aca03" }, { "dataPath": "params_shard_86.bin", "format": "raw-shard", "nbytes": 29268992, "records": [ { "name": "model.layers.26.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 0 }, { "name": "model.layers.26.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 8257536 }, { "name": "model.layers.26.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 9289728 }, { "name": "model.layers.26.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 15712256 }, { "name": "model.layers.27.input_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 16515072 }, { "name": "model.layers.27.mlp.down_proj.q_scale", "shape": [ 3584, 592 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 4243456, "byteOffset": 16522240 }, { "name": "model.layers.27.mlp.gate_up_proj.q_scale", "shape": [ 37888, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 8486912, "byteOffset": 20765696 }, { "name": "model.layers.27.post_attention_layernorm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 29252608 }, { "name": "model.layers.27.self_attn.c_attn.bias", "shape": [ 4608 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 9216, "byteOffset": 29259776 } ], "md5sum": "d5b1152abf7fd09809a43f255cf7a24d" }, { "dataPath": "params_shard_87.bin", "format": "raw-shard", "nbytes": 16522240, "records": [ { "name": "model.layers.27.self_attn.c_attn.q_weight", "shape": [ 4608, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 8257536, "byteOffset": 0 }, { "name": "model.layers.27.self_attn.c_attn.q_scale", "shape": [ 4608, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 1032192, "byteOffset": 8257536 }, { "name": "model.layers.27.self_attn.o_proj.q_weight", "shape": [ 3584, 448 ], "dtype": "uint32", "format": "f32-to-bf16", "nbytes": 6422528, "byteOffset": 9289728 }, { "name": "model.layers.27.self_attn.o_proj.q_scale", "shape": [ 3584, 112 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 802816, "byteOffset": 15712256 }, { "name": "model.norm.weight", "shape": [ 3584 ], "dtype": "float16", "format": "f32-to-bf16", "nbytes": 7168, "byteOffset": 16515072 } ], "md5sum": "6b2ea4578056dcac7e841100d3653899" } ] }