diff --git "a/ndarray-cache.json" "b/ndarray-cache.json" new file mode 100644--- /dev/null +++ "b/ndarray-cache.json" @@ -0,0 +1,3129 @@ +{ + "metadata": { + "ParamSize": 195, + "ParamBytes": 7642159104.0, + "BitsPerParam": 16.0 + }, + "records": [ + { + "dataPath": "params_shard_0.bin", + "format": "raw-shard", + "nbytes": 197001216, + "records": [ + { + "name": "lm_head.weight", + "shape": [ + 32064, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 197001216, + "byteOffset": 0 + } + ], + "md5sum": "8c2d85352caa7c9a98d74a9ee3cf6a94" + }, + { + "dataPath": "params_shard_1.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.21.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "fbdfcef1bc44437d1e2dd389140c2334" + }, + { + "dataPath": "params_shard_2.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.21.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "b14efca889e2938bd0dd70bf01f3e03a" + }, + { + "dataPath": "params_shard_3.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.21.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "c1a49fa4938c39315cb4cdcc5c2d7d49" + }, + { + "dataPath": "params_shard_4.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.22.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "a2239c27ad68b6f533a7102d82527878" + }, + { + "dataPath": "params_shard_5.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.22.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "c56bee40a05fbd4322eade65122c07f3" + }, + { + "dataPath": "params_shard_6.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.22.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "bfa63c8cb8115ffdfe3d0e9141eee648" + }, + { + "dataPath": "params_shard_7.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.23.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "83521bd8fbce11184521d6ab41a29cea" + }, + { + "dataPath": "params_shard_8.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.23.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "2ea3121970667f0f7645c5e728237748" + }, + { + "dataPath": "params_shard_9.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.23.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "1934b976730e3bd2abf6d9a8b72529d8" + }, + { + "dataPath": "params_shard_10.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.23.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "1cf17bdee74a90757820d2d22aabecf7" + }, + { + "dataPath": "params_shard_11.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.24.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "03112ed394c45b3446d495aeb5cbdff0" + }, + { + "dataPath": "params_shard_12.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.24.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "49188904aa2c30790064abfd8097672c" + }, + { + "dataPath": "params_shard_13.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.24.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "f42a37069ae388cfdaf2d09833d82244" + }, + { + "dataPath": "params_shard_14.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.24.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "30c14cf99aff89957ec7da295c4ab632" + }, + { + "dataPath": "params_shard_15.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.25.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "b57a23df765fd830804df014adca71e7" + }, + { + "dataPath": "params_shard_16.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.25.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "6a8895d6b1ccd90c8e91f75da70bdd3e" + }, + { + "dataPath": "params_shard_17.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.25.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "66114f0757c09a36c7ffd636719cc7b3" + }, + { + "dataPath": "params_shard_18.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.25.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "f9ba7dd1a9ffe04b85583439d91d551a" + }, + { + "dataPath": "params_shard_19.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.26.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "71e79d1578fd6ee4e4524679fb1276ff" + }, + { + "dataPath": "params_shard_20.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.26.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "5dc9d0557437fcf6e561ee3951baadae" + }, + { + "dataPath": "params_shard_21.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.26.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "297e4f56ec4f407b4e9546c012b09775" + }, + { + "dataPath": "params_shard_22.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.26.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "2bf6b964d4c2688cc13183649276342a" + }, + { + "dataPath": "params_shard_23.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.27.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "17b63513014e546d7b672228b045718c" + }, + { + "dataPath": "params_shard_24.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.27.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "487b046eee7038d7046ba4740a5ca521" + }, + { + "dataPath": "params_shard_25.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.27.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "30b04b01a54e57beba23e7fe29957c87" + }, + { + "dataPath": "params_shard_26.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.27.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "00ed36d8f6e889b0567ff936972055bb" + }, + { + "dataPath": "params_shard_27.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.28.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "d790288dcc53da10b9ee96e9a80571be" + }, + { + "dataPath": "params_shard_28.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.28.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "8625bcda8d93dc0529984087f07cccb8" + }, + { + "dataPath": "params_shard_29.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.28.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "ab0e01373b835f596283764432bdcea9" + }, + { + "dataPath": "params_shard_30.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.28.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "3b373274561aeb974d849cc01c2850b0" + }, + { + "dataPath": "params_shard_31.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.29.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "ac9774e76732119d81106d74f7f6ee52" + }, + { + "dataPath": "params_shard_32.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.29.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "47013b7213be9e785d22bdaa10590ca4" + }, + { + "dataPath": "params_shard_33.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.29.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "aaf205e3e3798acd2b6f1bc92d94d6de" + }, + { + "dataPath": "params_shard_34.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.29.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "d3b2899e4379328af50315fe46f9e144" + }, + { + "dataPath": "params_shard_35.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.30.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "d5604e270a26008514d6f07651520d7b" + }, + { + "dataPath": "params_shard_36.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.30.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "c21012a11160161b3f16436d5e8fbcea" + }, + { + "dataPath": "params_shard_37.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.30.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "99ac28b15ff15c0e7dcfada05307ac89" + }, + { + "dataPath": "params_shard_38.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.30.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "225394d73feea900325e934fd3a1c0b4" + }, + { + "dataPath": "params_shard_39.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.31.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "9d287bb8e3cf1f29df30aefc9c027fd2" + }, + { + "dataPath": "params_shard_40.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.31.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "263c716606e3a2afc30802ef6483db82" + }, + { + "dataPath": "params_shard_41.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.31.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "f0ecb725e774085ffd140fd5df0089e6" + }, + { + "dataPath": "params_shard_42.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.31.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "191fa72b252485f0a9db714e8a2fa3e5" + }, + { + "dataPath": "params_shard_43.bin", + "format": "raw-shard", + "nbytes": 197001216, + "records": [ + { + "name": "transformer.embd.weight", + "shape": [ + 32064, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 197001216, + "byteOffset": 0 + } + ], + "md5sum": "6f9e82ee16bf9e48202971f1bc7faddf" + }, + { + "dataPath": "params_shard_44.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.0.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "6d31c842666fb35f49c2826e9469172e" + }, + { + "dataPath": "params_shard_45.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.0.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "caf606342f712b439b4540b30b4d4dd8" + }, + { + "dataPath": "params_shard_46.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.0.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "3cbc63d65af31aeeb5f0ff6d21b05d91" + }, + { + "dataPath": "params_shard_47.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.0.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "8011494667ca20936cfbd3fe057f030f" + }, + { + "dataPath": "params_shard_48.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.1.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "d58dba16b580f460e040860a3be560d7" + }, + { + "dataPath": "params_shard_49.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.1.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "e3851f228e9eeea429ffdb5bfcf65712" + }, + { + "dataPath": "params_shard_50.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.1.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "36a299bba3d83d4b09fe38cea015c920" + }, + { + "dataPath": "params_shard_51.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.1.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "a2c6ca04c30deb8ca3dd477f9b070990" + }, + { + "dataPath": "params_shard_52.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.10.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "cc7e87530395bcaae2bbaaf14cdccf01" + }, + { + "dataPath": "params_shard_53.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.10.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "c39dc653bd8e615ccf216eafefa87cc9" + }, + { + "dataPath": "params_shard_54.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.10.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "46da89076ea5cf5e396f3e5b886af15e" + }, + { + "dataPath": "params_shard_55.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.10.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "dc18d2f273cd86d9bb06d7dcae88c221" + }, + { + "dataPath": "params_shard_56.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.11.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "36bc078291c38abe52684ab9eae3a086" + }, + { + "dataPath": "params_shard_57.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.11.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "00b40de1e12839f2fafaf0742585be28" + }, + { + "dataPath": "params_shard_58.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.11.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "d8bba7fe7d417dea96d3c791522604d6" + }, + { + "dataPath": "params_shard_59.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.11.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "652f758c4908ca7d5a76a3d241bb83a0" + }, + { + "dataPath": "params_shard_60.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.12.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "8cf3e143a89cbf61c8620a66b6446075" + }, + { + "dataPath": "params_shard_61.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.12.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "9101ba07812357ac26670e4b6a07dadf" + }, + { + "dataPath": "params_shard_62.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.12.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "756a1a808db0ed7583500bc0a2b54008" + }, + { + "dataPath": "params_shard_63.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.12.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "42262b1d88e9d9d49b33fda8283cb753" + }, + { + "dataPath": "params_shard_64.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.13.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "3bd7a6b66767bd6ebabcbeaa81cd0d15" + }, + { + "dataPath": "params_shard_65.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.13.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "dae0535b359e750b5168e08edaab2f0a" + }, + { + "dataPath": "params_shard_66.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.13.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "71fd9d016830fe9a2e62c82f4727b55f" + }, + { + "dataPath": "params_shard_67.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.13.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "bdac4468f08f6d325910744e53a73334" + }, + { + "dataPath": "params_shard_68.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.14.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "41331cffbf45ca84270d7f85134ece7c" + }, + { + "dataPath": "params_shard_69.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.14.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "22c314f049027022e97d39b3bd9e56fe" + }, + { + "dataPath": "params_shard_70.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.14.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "d51c88b65e6d2f35d7bd16e4ec04136b" + }, + { + "dataPath": "params_shard_71.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.14.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "72e28da4e109231f4a7466c4eec109cb" + }, + { + "dataPath": "params_shard_72.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.15.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "96c63673e9f0a437c157c676b2af5ee9" + }, + { + "dataPath": "params_shard_73.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.15.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "5558186b76ecb44c6cf7156980cb8978" + }, + { + "dataPath": "params_shard_74.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.15.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "79dc6596ac0f78f6f8ebad933ecb50fd" + }, + { + "dataPath": "params_shard_75.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.15.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "0b3553b64fa2313e345ee6baa3c258bb" + }, + { + "dataPath": "params_shard_76.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.16.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "e51ca07d5edd7aaae0d992b4e8dcfb21" + }, + { + "dataPath": "params_shard_77.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.16.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "7ee0335455612128b3b9171a15700bf2" + }, + { + "dataPath": "params_shard_78.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.16.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "18c35ebe336c1a121b23b1b1e4084a25" + }, + { + "dataPath": "params_shard_79.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.16.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "946796a5a531e3372a9eb3a0c7549430" + }, + { + "dataPath": "params_shard_80.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.17.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "93f0d0f857c235dfeb5ffdc9fab7351a" + }, + { + "dataPath": "params_shard_81.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.17.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "c59ecdf2dd9d347fedc9e5708f624d3c" + }, + { + "dataPath": "params_shard_82.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.17.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "335d4f2d9379b41144983b72b18a52ea" + }, + { + "dataPath": "params_shard_83.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.17.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "ad8afa5fe6d4634adadd2c9e8b2f069c" + }, + { + "dataPath": "params_shard_84.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.18.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "d458ff2536af888debbf4a0018dd1767" + }, + { + "dataPath": "params_shard_85.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.18.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "e7796e4854cdc538e01d7112ab8a5179" + }, + { + "dataPath": "params_shard_86.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.18.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "9df18282bd08212a8ae8fad347721ce2" + }, + { + "dataPath": "params_shard_87.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.18.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "0763aaa7da6da824f21db13a1bdd3832" + }, + { + "dataPath": "params_shard_88.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.19.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "a49f64fd634619b2ea22b320f9a65720" + }, + { + "dataPath": "params_shard_89.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.19.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "9bc101e6065af554c1f77e18793af166" + }, + { + "dataPath": "params_shard_90.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.19.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "b2e650b00c4be4a2699e1f4426c1d92f" + }, + { + "dataPath": "params_shard_91.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.19.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "9454b1638ae97f3d95aa11e6c5368d82" + }, + { + "dataPath": "params_shard_92.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.2.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "36232340846a5ed879134aa5a8336a0e" + }, + { + "dataPath": "params_shard_93.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.2.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "2c848444240798a2c8f998ecc0e7444b" + }, + { + "dataPath": "params_shard_94.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.2.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "5b0642334e6411b70e7cbfc0a203ead3" + }, + { + "dataPath": "params_shard_95.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.2.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "6b0e054b7f54523751b04449c6d2f9c7" + }, + { + "dataPath": "params_shard_96.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.20.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "74a5476d9d22d264626d5b9cbcddcb27" + }, + { + "dataPath": "params_shard_97.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.20.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "a1eee72b43f1c814da6b1e240aebd8fb" + }, + { + "dataPath": "params_shard_98.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.20.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "9274cf6c8ba095bebd1fd97af1a2e642" + }, + { + "dataPath": "params_shard_99.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.20.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "00b3cc2bc5d7dc8a7063579ecafeca01" + }, + { + "dataPath": "params_shard_100.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.21.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "acba950f76ad24093ed04f28c465d198" + }, + { + "dataPath": "params_shard_101.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.3.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "d8e4ac1c861de551f3345bc45c4a9745" + }, + { + "dataPath": "params_shard_102.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.3.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "2a020772f91756e032ebc7a920a2ece2" + }, + { + "dataPath": "params_shard_103.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.3.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "37a98cccee5ea8f0591dfdd76dbf3adb" + }, + { + "dataPath": "params_shard_104.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.3.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "67ad1c4d1ae5fa7c2385bedf20c8b73f" + }, + { + "dataPath": "params_shard_105.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.4.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "fe21c57ec308833254a7ab4b0ebbafe3" + }, + { + "dataPath": "params_shard_106.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.4.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "b3283f70b20bbe2237821ae571184a9a" + }, + { + "dataPath": "params_shard_107.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.4.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "5576fa782022a5c40c568017ba734d99" + }, + { + "dataPath": "params_shard_108.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.4.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "2920b6adb22a73082d3ac2f8741a974d" + }, + { + "dataPath": "params_shard_109.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.5.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "ffdf79001044f23701e18213bf76f0bd" + }, + { + "dataPath": "params_shard_110.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.5.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "715c79ed4715fb3e6d8d2d79c5f56b5e" + }, + { + "dataPath": "params_shard_111.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.5.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "654c5931d95b895a9a55ff2afbc699a5" + }, + { + "dataPath": "params_shard_112.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.5.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "f8eba59f4c24e5d784248fa5d18a42d8" + }, + { + "dataPath": "params_shard_113.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.6.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "2c66892223944f2ec60aaf69d34d2e11" + }, + { + "dataPath": "params_shard_114.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.6.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "37cd987183ccc430ce35b48fa7f05db3" + }, + { + "dataPath": "params_shard_115.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.6.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "fcf4c68aae219083c5b13dd54bb6a0f7" + }, + { + "dataPath": "params_shard_116.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.6.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "cfff45bedd5d3f9c1acdec8706ddfd12" + }, + { + "dataPath": "params_shard_117.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.7.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "7a908227d9ab71d3e4f605e44a4c0b84" + }, + { + "dataPath": "params_shard_118.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.7.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "a63968773cbb992bdeb65f804200f966" + }, + { + "dataPath": "params_shard_119.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.7.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "09a77c5c5de700053b2fd325baee2e3e" + }, + { + "dataPath": "params_shard_120.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.7.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "1331913d0a139cc0ea7f3c2ef19a1854" + }, + { + "dataPath": "params_shard_121.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.8.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "d0efc02d3f2d395960cc59a6a29941dd" + }, + { + "dataPath": "params_shard_122.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.8.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "515287320d750ddef90a2c26d59b45cc" + }, + { + "dataPath": "params_shard_123.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.8.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "48386e64e7bc21c2c6d9ec1efc5d348c" + }, + { + "dataPath": "params_shard_124.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.8.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "fcbd7dd1b3d229bbe556ed019dbb3272" + }, + { + "dataPath": "params_shard_125.bin", + "format": "raw-shard", + "nbytes": 50331648, + "records": [ + { + "name": "transformer.h.9.mlp.down_proj.weight", + "shape": [ + 3072, + 8192 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 50331648, + "byteOffset": 0 + } + ], + "md5sum": "d46cfb6c6df8baa7173adcb195a45609" + }, + { + "dataPath": "params_shard_126.bin", + "format": "raw-shard", + "nbytes": 100663296, + "records": [ + { + "name": "transformer.h.9.mlp.gate_up_proj.weight", + "shape": [ + 16384, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 100663296, + "byteOffset": 0 + } + ], + "md5sum": "1e6c430b422fc400e63d33c45f7bd44b" + }, + { + "dataPath": "params_shard_127.bin", + "format": "raw-shard", + "nbytes": 18874368, + "records": [ + { + "name": "transformer.h.9.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 0 + } + ], + "md5sum": "48d9d3fdff5e72ecf1569b8eec89f3e3" + }, + { + "dataPath": "params_shard_128.bin", + "format": "raw-shard", + "nbytes": 56623104, + "records": [ + { + "name": "transformer.h.9.mixer.qkv_proj.weight", + "shape": [ + 9216, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 56623104, + "byteOffset": 0 + } + ], + "md5sum": "d80713d97db0fe2e884ae7bc0d611235" + }, + { + "dataPath": "params_shard_129.bin", + "format": "raw-shard", + "nbytes": 19273728, + "records": [ + { + "name": "transformer.h.21.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 0 + }, + { + "name": "transformer.h.21.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 6144 + }, + { + "name": "transformer.h.22.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 12288 + }, + { + "name": "transformer.h.22.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18432 + }, + { + "name": "transformer.h.22.mixer.out_proj.weight", + "shape": [ + 3072, + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 18874368, + "byteOffset": 24576 + }, + { + "name": "transformer.h.23.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18898944 + }, + { + "name": "transformer.h.23.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18905088 + }, + { + "name": "transformer.h.24.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18911232 + }, + { + "name": "transformer.h.24.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18917376 + }, + { + "name": "transformer.h.25.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18923520 + }, + { + "name": "transformer.h.25.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18929664 + }, + { + "name": "transformer.h.26.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18935808 + }, + { + "name": "transformer.h.26.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18941952 + }, + { + "name": "transformer.h.27.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18948096 + }, + { + "name": "transformer.h.27.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18954240 + }, + { + "name": "transformer.h.28.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18960384 + }, + { + "name": "transformer.h.28.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18966528 + }, + { + "name": "transformer.h.29.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18972672 + }, + { + "name": "transformer.h.29.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18978816 + }, + { + "name": "transformer.h.30.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18984960 + }, + { + "name": "transformer.h.30.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18991104 + }, + { + "name": "transformer.h.31.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 18997248 + }, + { + "name": "transformer.h.31.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19003392 + }, + { + "name": "transformer.norm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19009536 + }, + { + "name": "transformer.h.0.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19015680 + }, + { + "name": "transformer.h.0.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19021824 + }, + { + "name": "transformer.h.1.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19027968 + }, + { + "name": "transformer.h.1.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19034112 + }, + { + "name": "transformer.h.10.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19040256 + }, + { + "name": "transformer.h.10.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19046400 + }, + { + "name": "transformer.h.11.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19052544 + }, + { + "name": "transformer.h.11.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19058688 + }, + { + "name": "transformer.h.12.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19064832 + }, + { + "name": "transformer.h.12.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19070976 + }, + { + "name": "transformer.h.13.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19077120 + }, + { + "name": "transformer.h.13.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19083264 + }, + { + "name": "transformer.h.14.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19089408 + }, + { + "name": "transformer.h.14.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19095552 + }, + { + "name": "transformer.h.15.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19101696 + }, + { + "name": "transformer.h.15.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19107840 + }, + { + "name": "transformer.h.16.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19113984 + }, + { + "name": "transformer.h.16.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19120128 + }, + { + "name": "transformer.h.17.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19126272 + }, + { + "name": "transformer.h.17.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19132416 + }, + { + "name": "transformer.h.18.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19138560 + }, + { + "name": "transformer.h.18.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19144704 + }, + { + "name": "transformer.h.19.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19150848 + }, + { + "name": "transformer.h.19.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19156992 + }, + { + "name": "transformer.h.2.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19163136 + }, + { + "name": "transformer.h.2.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19169280 + }, + { + "name": "transformer.h.20.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19175424 + }, + { + "name": "transformer.h.20.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19181568 + }, + { + "name": "transformer.h.3.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19187712 + }, + { + "name": "transformer.h.3.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19193856 + }, + { + "name": "transformer.h.4.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19200000 + }, + { + "name": "transformer.h.4.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19206144 + }, + { + "name": "transformer.h.5.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19212288 + }, + { + "name": "transformer.h.5.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19218432 + }, + { + "name": "transformer.h.6.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19224576 + }, + { + "name": "transformer.h.6.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19230720 + }, + { + "name": "transformer.h.7.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19236864 + }, + { + "name": "transformer.h.7.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19243008 + }, + { + "name": "transformer.h.8.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19249152 + }, + { + "name": "transformer.h.8.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19255296 + }, + { + "name": "transformer.h.9.ln.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19261440 + }, + { + "name": "transformer.h.9.post_attention_layernorm.weight", + "shape": [ + 3072 + ], + "dtype": "float16", + "format": "f32-to-bf16", + "nbytes": 6144, + "byteOffset": 19267584 + } + ], + "md5sum": "7d90cc7bee5c8a5aa80ac6fca9f03b61" + } + ] +} \ No newline at end of file