diff --git a/model-00001-of-00059.safetensors b/model-00001-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..b89c07ad3c2df7926ecc8b1027cc1cc43474625a --- /dev/null +++ b/model-00001-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0f52e6fd4c769858cf48e6ae3e75d777b36b406c7e0729cdc4ea1536c6b07942 +size 4998737424 diff --git a/model-00002-of-00059.safetensors b/model-00002-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..82dfd97dc770a039cbc880f0ebc71f4afd978db4 --- /dev/null +++ b/model-00002-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5a3f91c7d5477ae60bb50d41c4e6b9372dda3e91a3129eba70c2afb8ff4f7e4a +size 4806799120 diff --git a/model-00003-of-00059.safetensors b/model-00003-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..58dd89a37f546b62b241d9a5bdc98aa42adec071 --- /dev/null +++ b/model-00003-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6ba6f3bb2e2e8b5d8959b61e6751d91641ef8e354c0c1a67f6d8577343472e81 +size 4806799120 diff --git a/model-00004-of-00059.safetensors b/model-00004-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..ffbe64a32b6271f2f1c015cb111604fc9b1c5a50 --- /dev/null +++ b/model-00004-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7cf3e48e7855ee38cc0e226ee7492f3d8e1b2c7bcf545b49752c6e1724487f24 +size 4806799120 diff --git a/model-00005-of-00059.safetensors b/model-00005-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..438ec2b366645a55c3cc54e34be6ac7921b26951 --- /dev/null +++ b/model-00005-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3b811038a648f9dbd2dead3f6eed998290b1116e45a4a8e9688767f8491387be +size 4806799120 diff --git a/model-00006-of-00059.safetensors b/model-00006-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..59fd9215fef29dfd794776624db9d133e15c8681 --- /dev/null +++ b/model-00006-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:63a5af5e2f1c8f052659bc9a79fbde8ff550bb4c957768c04dddfbf254802a74 +size 4806799120 diff --git a/model-00007-of-00059.safetensors b/model-00007-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..26c6412f7f133189d9db1fbfff561c98856bec7c --- /dev/null +++ b/model-00007-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e3a33eba3ceaa013d47d89b925bbba771d19ff3b1a2eda829e2da59ba0d46de4 +size 4806799120 diff --git a/model-00008-of-00059.safetensors b/model-00008-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..4bedb376595911b68310c1e3f1e8b04abbedd0d6 --- /dev/null +++ b/model-00008-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:612afec4ecde04b5aa4056f174012aee3ef0111336a717b853e4dde9f03c685a +size 4806799120 diff --git a/model-00009-of-00059.safetensors b/model-00009-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..9679e76a200293704396447672bedc6e8eeacec5 --- /dev/null +++ b/model-00009-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9e3e8c37a777daf1156dc3b33a24fbc1ffc301ce6509876b0dc10bacd903def5 +size 4806799120 diff --git a/model-00010-of-00059.safetensors b/model-00010-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..2bd1951efca71a215b87e94f59e16615df52b5a3 --- /dev/null +++ b/model-00010-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a2ddd1e4ad72f2acee4f741d0a6a9836c3091b9ca380f3780357dfc8281a470c +size 4806799120 diff --git a/model-00011-of-00059.safetensors b/model-00011-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..847fa5f3b0cc1238bd6e865f0acaf5fa3563443d --- /dev/null +++ b/model-00011-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f856902e058b8ebc14add8809661501b5cd4be7d4ea483c10641f247fcdc10e8 +size 4806799136 diff --git a/model-00012-of-00059.safetensors b/model-00012-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..4eae9907f45adc4f3b2075d5d09d8eddfaa7c148 --- /dev/null +++ b/model-00012-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:698354b30e6777e6e0205014451035752d6d30a43e012783f029b39167b0a233 +size 4806799152 diff --git a/model-00013-of-00059.safetensors b/model-00013-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..35d3ca01a4b6297c45501f8ea9c395921236dd2e --- /dev/null +++ b/model-00013-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ef69a980b472038e56cd23aa094a875e96da8cc2cb676560798964a95161b71f +size 4806799152 diff --git a/model-00014-of-00059.safetensors b/model-00014-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..f0146dff9a92dffaadbffda1320b088ae5be43f5 --- /dev/null +++ b/model-00014-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c173d71254eda60eb9ddad7cf7eab41960286e672032b848bc6d8b16a7d94065 +size 4806799152 diff --git a/model-00015-of-00059.safetensors b/model-00015-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..2baab9dde16822671ed5635f26e879261149f416 --- /dev/null +++ b/model-00015-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:eee3517c068604a5f3248c21912d6ba32844e043088ad561bea5224206a1b8c4 +size 4806799152 diff --git a/model-00016-of-00059.safetensors b/model-00016-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..4ea544e97339efaa8c47da5c6fc8ec7c85e12a37 --- /dev/null +++ b/model-00016-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cafa78bfdcea8cbc3fff4979059b242bce68988ebf2a4a56ea62642f8a7446fb +size 4806799152 diff --git a/model-00017-of-00059.safetensors b/model-00017-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..13daa65af603455e02fd5c8b4bb07da8b419c0cd --- /dev/null +++ b/model-00017-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:354898b9f08c86a83dc4ab8d992655df2d8bbdb2156aa4d4cade5b91440d65ef +size 4806799152 diff --git a/model-00018-of-00059.safetensors b/model-00018-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..9cb9b7b5c3c5627966177b548535c2eff4a1fefd --- /dev/null +++ b/model-00018-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7e05cc575a9025c921084683e2c78782bc07011e6272d9a9bce10daf36f5353e +size 4806799152 diff --git a/model-00019-of-00059.safetensors b/model-00019-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..6560a75c2f29a015a09b656c1bb083e5ae3e3991 --- /dev/null +++ b/model-00019-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b6ea5b54dbcb9f48db7aeab8d994dd555b7d1f71b175b19fea03a94deda82661 +size 4806799152 diff --git a/model-00020-of-00059.safetensors b/model-00020-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..14a5b263415d4ed8ced409fdc9302d182b1210d3 --- /dev/null +++ b/model-00020-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:76e382df0e331af8d0583e9d95116d42a6db00f16c74dbf6e71b982759651ba9 +size 4806799152 diff --git a/model-00021-of-00059.safetensors b/model-00021-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0db8307cf2ef1964ba21b982e9f4ba2686a4dd08 --- /dev/null +++ b/model-00021-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cab60641df319f25fc03048d82c360eff5303d0db4b93209fae99bba4db117f7 +size 4806799152 diff --git a/model-00022-of-00059.safetensors b/model-00022-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..71a2ff3b726e19e66d20e57d8c1a515302720829 --- /dev/null +++ b/model-00022-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bfe228acd775279b56aebf657a7fa5afce881a6296f74dd0e6164f0d3e238d85 +size 4806799152 diff --git a/model-00023-of-00059.safetensors b/model-00023-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..54e467019cc7b80d5be1d49c5c9006bd12b4aa09 --- /dev/null +++ b/model-00023-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5b0a2e1fb46ae238a9192ca1f8c699e1d88898f6c16486913c8cac487d4f12e0 +size 4806799152 diff --git a/model-00024-of-00059.safetensors b/model-00024-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..bf888e9d864bf0a472ee16429c56db9e95654bd3 --- /dev/null +++ b/model-00024-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ce118792b55a2c73efc45a06e4bf48382f6d0feb287043462c075d7ea1d8dac5 +size 4932529864 diff --git a/model-00025-of-00059.safetensors b/model-00025-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0ba13101f5af14fc5e68ac307c29086d97859dca --- /dev/null +++ b/model-00025-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8c08d13b466912a7073b2f62a09e1e87a78f3cdf9d5e84eafc867f5a67a1928a +size 4995542848 diff --git a/model-00026-of-00059.safetensors b/model-00026-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..80413745a2161e4b53d4b065257416ad843d7cb9 --- /dev/null +++ b/model-00026-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6597c73dc2ebf8ec2d1182211cd4c25c1c859a5b5065c94799fc36165e3bde18 +size 4995542848 diff --git a/model-00027-of-00059.safetensors b/model-00027-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..431574f725f300445cfce301e155363619c33d0a --- /dev/null +++ b/model-00027-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6b189e18a819290a837a1d7f1f31d0946b6737901b5af0a6f0e2b4c11d490bf3 +size 4932628288 diff --git a/model-00028-of-00059.safetensors b/model-00028-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..69cdafca47a483e1dcdde13d5b32b07e8e9d4216 --- /dev/null +++ b/model-00028-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0cd1a2060973cc2364d0252b66652d8bf4323b18e10c5f648f06fa49bd4288fb +size 4806774344 diff --git a/model-00029-of-00059.safetensors b/model-00029-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..d9c4a6efa493f41b436533865ccaafcb7ffda000 --- /dev/null +++ b/model-00029-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:389a4b63c9c17312381f2260aa6d644688de3a10535990e4b6251df3bc04f66c +size 4806799144 diff --git a/model-00030-of-00059.safetensors b/model-00030-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..451bf748704dc179af87ee1af2fffacb8e1e5d23 --- /dev/null +++ b/model-00030-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:faaadc6c606b6c9be394da786e69b58fe0e858c2e0ddf327a3898c69b1e8bfde +size 4806799144 diff --git a/model-00031-of-00059.safetensors b/model-00031-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..ef51774c3d3864090dcc041d3462c955fbf34dcd --- /dev/null +++ b/model-00031-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1c0a366eda2795e9b13439ac7980917b9c3f983767730c24f312dc16fafd3f09 +size 4806799144 diff --git a/model-00032-of-00059.safetensors b/model-00032-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..08ed58d29070e299dcbc42ab1aefd8c70d62b4b8 --- /dev/null +++ b/model-00032-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e4d01bc12479a1860ad618e8933f03a39b690911ee3a0dae06b81671528f419b +size 4806799144 diff --git a/model-00033-of-00059.safetensors b/model-00033-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..6efc478aef62264bc9e692e92cb638af7345716a --- /dev/null +++ b/model-00033-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3e7c8bbbe9963a81e3dc20962b0e27f22d31aea9404e3385353a1c57044d8c00 +size 4806799152 diff --git a/model-00034-of-00059.safetensors b/model-00034-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..ebc08967d08f74c9280b95fe2c83720abed57358 --- /dev/null +++ b/model-00034-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c621dc9392fc359df8e810d15d3f2bd71ab38bb0e0013e9909b8972f153d3584 +size 4806799152 diff --git a/model-00035-of-00059.safetensors b/model-00035-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0d24f36b569314a1cac6daff556779b6b6520f34 --- /dev/null +++ b/model-00035-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:86dde6a6b3732f79f0872471e8e4f9f45c39f1d86eeef8cf9955b93d97b168f5 +size 4806799152 diff --git a/model-00036-of-00059.safetensors b/model-00036-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..9e8a66a549fb514cf9da5794d4ffd2086e82ef5d --- /dev/null +++ b/model-00036-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1226018b9d3ff76024c8f95444f5f1994fd77dfd3183682d5e957b1b99ad752f +size 4806799152 diff --git a/model-00037-of-00059.safetensors b/model-00037-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..12da7b5fa449cb04cf526fb7b4c19c4bfe31a051 --- /dev/null +++ b/model-00037-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8c0bd3e38d055c40551992c123e856ed2f91455e77c85ea75d64caa3db270753 +size 4806799152 diff --git a/model-00038-of-00059.safetensors b/model-00038-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..9ea79baf7bfc89be6cbebe3f5f13724ae4076864 --- /dev/null +++ b/model-00038-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f0ca048c75da48a5952b59235a76ead816779811377fe09840f9897f25a22813 +size 4806799152 diff --git a/model-00039-of-00059.safetensors b/model-00039-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..e9f96270f9b025ac97f44a4c1acf1a5933ac1959 --- /dev/null +++ b/model-00039-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b5fd6bfe4ee99014b4366f47f42066f9e6a76045fa0e7ae5ab4f567bae09338b +size 4806799152 diff --git a/model-00040-of-00059.safetensors b/model-00040-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..cd89ab2ba3e88c688ba6ce84d0584a2146ce2a4d --- /dev/null +++ b/model-00040-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9a53335a1fede6b98082f22003cf331fe288ed222dbf96060d2df8109ba3bb34 +size 4806799152 diff --git a/model-00041-of-00059.safetensors b/model-00041-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..f878e8c4118c30f3c2a7d8db98507afdee626c5a --- /dev/null +++ b/model-00041-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a3c87575fdae0c2acb1d08325e70aa8c117de36c81ee8cfd4e2a95d449cebf2c +size 4806799152 diff --git a/model-00042-of-00059.safetensors b/model-00042-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..895aae456b8c0fd2946e16c4f2b8758df4c7e2ca --- /dev/null +++ b/model-00042-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4c0a40f50db09ae81eba9d16210f34fd4bc913b233244879806a64569457b6dd +size 4806799152 diff --git a/model-00043-of-00059.safetensors b/model-00043-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..9d16be1a6d8cb9e1f047ce99d364b26b71f17812 --- /dev/null +++ b/model-00043-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d8e5c1331a2db40737855c5759cc9e34b313fe8b714d9e9f1ee0a3a8161e1883 +size 4806799152 diff --git a/model-00044-of-00059.safetensors b/model-00044-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..f75e1d7c0462a6c325d4a0f8a5ffc8e032a8f89c --- /dev/null +++ b/model-00044-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:867d24a93fe1dd0ee9980f4c3554c72383934faa6e2b27549cabbac71ee164e0 +size 4806799152 diff --git a/model-00045-of-00059.safetensors b/model-00045-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..767ac7cf55cf70b51dc2d1f4b402e745e8210e85 --- /dev/null +++ b/model-00045-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7f912dd273add5473e3648532427f505467a065be0da8ae278fe4f008f0a00cc +size 4806799152 diff --git a/model-00046-of-00059.safetensors b/model-00046-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..9e0ede74a4029713cfa3271943e9be18f8b32ff2 --- /dev/null +++ b/model-00046-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f12060b33bf49744178d4674b979fd41bbe7f48382b852878cfc02f1db730358 +size 4806799152 diff --git a/model-00047-of-00059.safetensors b/model-00047-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..7ae1a1f914a557ede4baee7280f284abca2b7fcf --- /dev/null +++ b/model-00047-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e5340de59956171882578698264375d5c6565abe3d60c0de475ee0706dd440ed +size 4806799152 diff --git a/model-00048-of-00059.safetensors b/model-00048-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..e14eb54319e16ee2bab9b8b14bdbee4912664ebe --- /dev/null +++ b/model-00048-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7a9fc6a5df101e4166444d96eb6686f9dab5e4afbe4ab91071a629fd431909b3 +size 4806799152 diff --git a/model-00049-of-00059.safetensors b/model-00049-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..b30d59684a113fdbad69696cd7a8aaabbd608478 --- /dev/null +++ b/model-00049-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4562eebed87184d994b8e1b105193881363255a98e0f9f4ffcdbda187f9eefc0 +size 4806799152 diff --git a/model-00050-of-00059.safetensors b/model-00050-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..33fa2e5e48a63842dd546ec351bc3f7e6c2a64f8 --- /dev/null +++ b/model-00050-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b02e73c4d7f6c8e212aa1bbe6291f681616b352a82417fffb3624525b8bd772e +size 4806799152 diff --git a/model-00051-of-00059.safetensors b/model-00051-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..204eaa28c4fa3dfe1a417f60ae4b74912ac6a592 --- /dev/null +++ b/model-00051-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4d90137ae99c7cd9487eb187dbb1a2b67af820a583d5a2dd54430fd2e8277357 +size 4806799152 diff --git a/model-00052-of-00059.safetensors b/model-00052-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..d7af0b343896230390612788e18b7b89f9b75f3f --- /dev/null +++ b/model-00052-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7c0d6fe105e8192f3822ba6a6e4e270049d5dad550bd7dcf463dabf5d4aedc46 +size 4932529864 diff --git a/model-00053-of-00059.safetensors b/model-00053-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..884b54ed894504f7a1c9f349a8c029ee09141561 --- /dev/null +++ b/model-00053-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0366b8df4599f8c20dc8805d53202476b9cb3d5d8f2c883522f0b438e5517d07 +size 4995542848 diff --git a/model-00054-of-00059.safetensors b/model-00054-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..e20676a1ed0741d6d1a2dfecfa85592d9e23c112 --- /dev/null +++ b/model-00054-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1bb6d8cc407e7363d9fa98ce93556871364a393d8e38544e561b17f02f37cdf2 +size 4995542848 diff --git a/model-00055-of-00059.safetensors b/model-00055-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..f8c5272652c7e3b3adf9e4444e5286830f491c4a --- /dev/null +++ b/model-00055-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:67ab152b370bd5f569adf312f59e574e97aa60d862d849088c043fe04f221532 +size 4932628288 diff --git a/model-00056-of-00059.safetensors b/model-00056-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..fd588355d6f3d19267ff6ce8cd47f5174c5854bb --- /dev/null +++ b/model-00056-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:31a106f4dee1f8bf0cb2662b0a0c07f669d9881b150eac6bad2732661e0b60f0 +size 4806774344 diff --git a/model-00057-of-00059.safetensors b/model-00057-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..701b0025ef0b87261af05e02e777ec1d9f5573c5 --- /dev/null +++ b/model-00057-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a468e0fb7beea11e9b9ad9163983ab7ba2f46cd9955decf1ddd87d8cc909ce68 +size 4806799144 diff --git a/model-00058-of-00059.safetensors b/model-00058-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..3d0c80780e4180a6e5fdedc59dcf54cae10cd48b --- /dev/null +++ b/model-00058-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c4fbad15c13eeb58df28d99e765a699fdc59ac9565884f22c3097d3548d3e831 +size 4806799144 diff --git a/model-00059-of-00059.safetensors b/model-00059-of-00059.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..54591422716df3eb111db23c4d32a7427ee71360 --- /dev/null +++ b/model-00059-of-00059.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3040fee72515c293c3f0a1280da841cd1f6aa86d2c0729ba600533f87dc55186 +size 997307200 diff --git a/model.safetensors.index.json b/model.safetensors.index.json new file mode 100644 index 0000000000000000000000000000000000000000..9f59e75a5127e27f7733ff00833e5e0ee4798bfa --- /dev/null +++ b/model.safetensors.index.json @@ -0,0 +1,1746 @@ +{ + "metadata": { + "total_size": 281241415680 + }, + "weight_map": { + "lm_head.weight": "model-00059-of-00059.safetensors", + "model.embed_tokens.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.0.w1.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.0.w2.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.0.w3.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.1.w1.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.1.w2.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.1.w3.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.2.w1.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.2.w2.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.2.w3.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.3.w1.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.3.w2.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.3.w3.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.4.w1.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.4.w2.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.4.w3.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.5.w1.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.5.w2.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.5.w3.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.6.w1.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.6.w2.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.6.w3.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.7.w1.weight": "model-00001-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.7.w2.weight": "model-00002-of-00059.safetensors", + "model.layers.0.block_sparse_moe.experts.7.w3.weight": "model-00002-of-00059.safetensors", + "model.layers.0.block_sparse_moe.gate.weight": "model-00001-of-00059.safetensors", + "model.layers.0.input_layernorm.weight": "model-00002-of-00059.safetensors", + "model.layers.0.post_attention_layernorm.weight": "model-00002-of-00059.safetensors", + "model.layers.0.self_attn.k_proj.weight": "model-00001-of-00059.safetensors", + "model.layers.0.self_attn.o_proj.weight": "model-00001-of-00059.safetensors", + "model.layers.0.self_attn.q_proj.weight": "model-00001-of-00059.safetensors", + "model.layers.0.self_attn.v_proj.weight": "model-00001-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.0.w1.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.0.w2.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.0.w3.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.1.w1.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.1.w2.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.1.w3.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.2.w1.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.2.w2.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.2.w3.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.3.w1.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.3.w2.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.3.w3.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.4.w1.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.4.w2.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.4.w3.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.5.w1.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.5.w2.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.5.w3.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.6.w1.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.6.w2.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.6.w3.weight": "model-00002-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.7.w1.weight": "model-00003-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.7.w2.weight": "model-00003-of-00059.safetensors", + "model.layers.1.block_sparse_moe.experts.7.w3.weight": "model-00003-of-00059.safetensors", + "model.layers.1.block_sparse_moe.gate.weight": "model-00002-of-00059.safetensors", + "model.layers.1.input_layernorm.weight": "model-00003-of-00059.safetensors", + "model.layers.1.post_attention_layernorm.weight": "model-00003-of-00059.safetensors", + "model.layers.1.self_attn.k_proj.weight": "model-00002-of-00059.safetensors", + "model.layers.1.self_attn.o_proj.weight": "model-00002-of-00059.safetensors", + "model.layers.1.self_attn.q_proj.weight": "model-00002-of-00059.safetensors", + "model.layers.1.self_attn.v_proj.weight": "model-00002-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.0.w1.weight": "model-00011-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.0.w2.weight": "model-00011-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.0.w3.weight": "model-00011-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.1.w1.weight": "model-00011-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.1.w2.weight": "model-00011-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.1.w3.weight": "model-00011-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.2.w1.weight": "model-00011-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.2.w2.weight": "model-00011-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.2.w3.weight": "model-00011-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.3.w1.weight": "model-00011-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.3.w2.weight": "model-00011-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.3.w3.weight": "model-00011-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.4.w1.weight": "model-00012-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.4.w2.weight": "model-00012-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.4.w3.weight": "model-00012-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.5.w1.weight": "model-00012-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.5.w2.weight": "model-00012-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.5.w3.weight": "model-00012-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.6.w1.weight": "model-00012-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.6.w2.weight": "model-00012-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.6.w3.weight": "model-00012-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.7.w1.weight": "model-00012-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.7.w2.weight": "model-00012-of-00059.safetensors", + "model.layers.10.block_sparse_moe.experts.7.w3.weight": "model-00012-of-00059.safetensors", + "model.layers.10.block_sparse_moe.gate.weight": "model-00011-of-00059.safetensors", + "model.layers.10.input_layernorm.weight": "model-00012-of-00059.safetensors", + "model.layers.10.post_attention_layernorm.weight": "model-00012-of-00059.safetensors", + "model.layers.10.self_attn.k_proj.weight": "model-00011-of-00059.safetensors", + "model.layers.10.self_attn.o_proj.weight": "model-00011-of-00059.safetensors", + "model.layers.10.self_attn.q_proj.weight": "model-00011-of-00059.safetensors", + "model.layers.10.self_attn.v_proj.weight": "model-00011-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.0.w1.weight": "model-00012-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.0.w2.weight": "model-00012-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.0.w3.weight": "model-00012-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.1.w1.weight": "model-00012-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.1.w2.weight": "model-00012-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.1.w3.weight": "model-00012-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.2.w1.weight": "model-00012-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.2.w2.weight": "model-00012-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.2.w3.weight": "model-00012-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.3.w1.weight": "model-00012-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.3.w2.weight": "model-00012-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.3.w3.weight": "model-00013-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.4.w1.weight": "model-00013-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.4.w2.weight": "model-00013-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.4.w3.weight": "model-00013-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.5.w1.weight": "model-00013-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.5.w2.weight": "model-00013-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.5.w3.weight": "model-00013-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.6.w1.weight": "model-00013-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.6.w2.weight": "model-00013-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.6.w3.weight": "model-00013-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.7.w1.weight": "model-00013-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.7.w2.weight": "model-00013-of-00059.safetensors", + "model.layers.11.block_sparse_moe.experts.7.w3.weight": "model-00013-of-00059.safetensors", + "model.layers.11.block_sparse_moe.gate.weight": "model-00012-of-00059.safetensors", + "model.layers.11.input_layernorm.weight": "model-00013-of-00059.safetensors", + "model.layers.11.post_attention_layernorm.weight": "model-00013-of-00059.safetensors", + "model.layers.11.self_attn.k_proj.weight": "model-00012-of-00059.safetensors", + "model.layers.11.self_attn.o_proj.weight": "model-00012-of-00059.safetensors", + "model.layers.11.self_attn.q_proj.weight": "model-00012-of-00059.safetensors", + "model.layers.11.self_attn.v_proj.weight": "model-00012-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.0.w1.weight": "model-00013-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.0.w2.weight": "model-00013-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.0.w3.weight": "model-00013-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.1.w1.weight": "model-00013-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.1.w2.weight": "model-00013-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.1.w3.weight": "model-00013-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.2.w1.weight": "model-00013-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.2.w2.weight": "model-00013-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.2.w3.weight": "model-00013-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.3.w1.weight": "model-00013-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.3.w2.weight": "model-00014-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.3.w3.weight": "model-00014-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.4.w1.weight": "model-00014-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.4.w2.weight": "model-00014-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.4.w3.weight": "model-00014-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.5.w1.weight": "model-00014-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.5.w2.weight": "model-00014-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.5.w3.weight": "model-00014-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.6.w1.weight": "model-00014-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.6.w2.weight": "model-00014-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.6.w3.weight": "model-00014-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.7.w1.weight": "model-00014-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.7.w2.weight": "model-00014-of-00059.safetensors", + "model.layers.12.block_sparse_moe.experts.7.w3.weight": "model-00014-of-00059.safetensors", + "model.layers.12.block_sparse_moe.gate.weight": "model-00013-of-00059.safetensors", + "model.layers.12.input_layernorm.weight": "model-00014-of-00059.safetensors", + "model.layers.12.post_attention_layernorm.weight": "model-00014-of-00059.safetensors", + "model.layers.12.self_attn.k_proj.weight": "model-00013-of-00059.safetensors", + "model.layers.12.self_attn.o_proj.weight": "model-00013-of-00059.safetensors", + "model.layers.12.self_attn.q_proj.weight": "model-00013-of-00059.safetensors", + "model.layers.12.self_attn.v_proj.weight": "model-00013-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.0.w1.weight": "model-00014-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.0.w2.weight": "model-00014-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.0.w3.weight": "model-00014-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.1.w1.weight": "model-00014-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.1.w2.weight": "model-00014-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.1.w3.weight": "model-00014-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.2.w1.weight": "model-00014-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.2.w2.weight": "model-00014-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.2.w3.weight": "model-00014-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.3.w1.weight": "model-00015-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.3.w2.weight": "model-00015-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.3.w3.weight": "model-00015-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.4.w1.weight": "model-00015-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.4.w2.weight": "model-00015-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.4.w3.weight": "model-00015-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.5.w1.weight": "model-00015-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.5.w2.weight": "model-00015-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.5.w3.weight": "model-00015-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.6.w1.weight": "model-00015-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.6.w2.weight": "model-00015-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.6.w3.weight": "model-00015-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.7.w1.weight": "model-00015-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.7.w2.weight": "model-00015-of-00059.safetensors", + "model.layers.13.block_sparse_moe.experts.7.w3.weight": "model-00015-of-00059.safetensors", + "model.layers.13.block_sparse_moe.gate.weight": "model-00014-of-00059.safetensors", + "model.layers.13.input_layernorm.weight": "model-00015-of-00059.safetensors", + "model.layers.13.post_attention_layernorm.weight": "model-00015-of-00059.safetensors", + "model.layers.13.self_attn.k_proj.weight": "model-00014-of-00059.safetensors", + "model.layers.13.self_attn.o_proj.weight": "model-00014-of-00059.safetensors", + "model.layers.13.self_attn.q_proj.weight": "model-00014-of-00059.safetensors", + "model.layers.13.self_attn.v_proj.weight": "model-00014-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.0.w1.weight": "model-00015-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.0.w2.weight": "model-00015-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.0.w3.weight": "model-00015-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.1.w1.weight": "model-00015-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.1.w2.weight": "model-00015-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.1.w3.weight": "model-00015-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.2.w1.weight": "model-00015-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.2.w2.weight": "model-00015-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.2.w3.weight": "model-00016-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.3.w1.weight": "model-00016-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.3.w2.weight": "model-00016-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.3.w3.weight": "model-00016-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.4.w1.weight": "model-00016-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.4.w2.weight": "model-00016-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.4.w3.weight": "model-00016-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.5.w1.weight": "model-00016-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.5.w2.weight": "model-00016-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.5.w3.weight": "model-00016-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.6.w1.weight": "model-00016-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.6.w2.weight": "model-00016-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.6.w3.weight": "model-00016-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.7.w1.weight": "model-00016-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.7.w2.weight": "model-00016-of-00059.safetensors", + "model.layers.14.block_sparse_moe.experts.7.w3.weight": "model-00016-of-00059.safetensors", + "model.layers.14.block_sparse_moe.gate.weight": "model-00015-of-00059.safetensors", + "model.layers.14.input_layernorm.weight": "model-00016-of-00059.safetensors", + "model.layers.14.post_attention_layernorm.weight": "model-00016-of-00059.safetensors", + "model.layers.14.self_attn.k_proj.weight": "model-00015-of-00059.safetensors", + "model.layers.14.self_attn.o_proj.weight": "model-00015-of-00059.safetensors", + "model.layers.14.self_attn.q_proj.weight": "model-00015-of-00059.safetensors", + "model.layers.14.self_attn.v_proj.weight": "model-00015-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.0.w1.weight": "model-00016-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.0.w2.weight": "model-00016-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.0.w3.weight": "model-00016-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.1.w1.weight": "model-00016-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.1.w2.weight": "model-00016-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.1.w3.weight": "model-00016-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.2.w1.weight": "model-00016-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.2.w2.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.2.w3.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.3.w1.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.3.w2.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.3.w3.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.4.w1.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.4.w2.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.4.w3.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.5.w1.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.5.w2.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.5.w3.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.6.w1.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.6.w2.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.6.w3.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.7.w1.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.7.w2.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.experts.7.w3.weight": "model-00017-of-00059.safetensors", + "model.layers.15.block_sparse_moe.gate.weight": "model-00016-of-00059.safetensors", + "model.layers.15.input_layernorm.weight": "model-00017-of-00059.safetensors", + "model.layers.15.post_attention_layernorm.weight": "model-00017-of-00059.safetensors", + "model.layers.15.self_attn.k_proj.weight": "model-00016-of-00059.safetensors", + "model.layers.15.self_attn.o_proj.weight": "model-00016-of-00059.safetensors", + "model.layers.15.self_attn.q_proj.weight": "model-00016-of-00059.safetensors", + "model.layers.15.self_attn.v_proj.weight": "model-00016-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.0.w1.weight": "model-00017-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.0.w2.weight": "model-00017-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.0.w3.weight": "model-00017-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.1.w1.weight": "model-00017-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.1.w2.weight": "model-00017-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.1.w3.weight": "model-00017-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.2.w1.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.2.w2.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.2.w3.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.3.w1.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.3.w2.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.3.w3.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.4.w1.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.4.w2.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.4.w3.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.5.w1.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.5.w2.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.5.w3.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.6.w1.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.6.w2.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.6.w3.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.7.w1.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.7.w2.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.experts.7.w3.weight": "model-00018-of-00059.safetensors", + "model.layers.16.block_sparse_moe.gate.weight": "model-00017-of-00059.safetensors", + "model.layers.16.input_layernorm.weight": "model-00018-of-00059.safetensors", + "model.layers.16.post_attention_layernorm.weight": "model-00018-of-00059.safetensors", + "model.layers.16.self_attn.k_proj.weight": "model-00017-of-00059.safetensors", + "model.layers.16.self_attn.o_proj.weight": "model-00017-of-00059.safetensors", + "model.layers.16.self_attn.q_proj.weight": "model-00017-of-00059.safetensors", + "model.layers.16.self_attn.v_proj.weight": "model-00017-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.0.w1.weight": "model-00018-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.0.w2.weight": "model-00018-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.0.w3.weight": "model-00018-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.1.w1.weight": "model-00018-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.1.w2.weight": "model-00018-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.1.w3.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.2.w1.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.2.w2.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.2.w3.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.3.w1.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.3.w2.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.3.w3.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.4.w1.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.4.w2.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.4.w3.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.5.w1.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.5.w2.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.5.w3.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.6.w1.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.6.w2.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.6.w3.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.7.w1.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.7.w2.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.experts.7.w3.weight": "model-00019-of-00059.safetensors", + "model.layers.17.block_sparse_moe.gate.weight": "model-00018-of-00059.safetensors", + "model.layers.17.input_layernorm.weight": "model-00019-of-00059.safetensors", + "model.layers.17.post_attention_layernorm.weight": "model-00019-of-00059.safetensors", + "model.layers.17.self_attn.k_proj.weight": "model-00018-of-00059.safetensors", + "model.layers.17.self_attn.o_proj.weight": "model-00018-of-00059.safetensors", + "model.layers.17.self_attn.q_proj.weight": "model-00018-of-00059.safetensors", + "model.layers.17.self_attn.v_proj.weight": "model-00018-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.0.w1.weight": "model-00019-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.0.w2.weight": "model-00019-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.0.w3.weight": "model-00019-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.1.w1.weight": "model-00019-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.1.w2.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.1.w3.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.2.w1.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.2.w2.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.2.w3.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.3.w1.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.3.w2.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.3.w3.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.4.w1.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.4.w2.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.4.w3.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.5.w1.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.5.w2.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.5.w3.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.6.w1.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.6.w2.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.6.w3.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.7.w1.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.7.w2.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.experts.7.w3.weight": "model-00020-of-00059.safetensors", + "model.layers.18.block_sparse_moe.gate.weight": "model-00019-of-00059.safetensors", + "model.layers.18.input_layernorm.weight": "model-00020-of-00059.safetensors", + "model.layers.18.post_attention_layernorm.weight": "model-00020-of-00059.safetensors", + "model.layers.18.self_attn.k_proj.weight": "model-00019-of-00059.safetensors", + "model.layers.18.self_attn.o_proj.weight": "model-00019-of-00059.safetensors", + "model.layers.18.self_attn.q_proj.weight": "model-00019-of-00059.safetensors", + "model.layers.18.self_attn.v_proj.weight": "model-00019-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.0.w1.weight": "model-00020-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.0.w2.weight": "model-00020-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.0.w3.weight": "model-00020-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.1.w1.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.1.w2.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.1.w3.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.2.w1.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.2.w2.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.2.w3.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.3.w1.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.3.w2.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.3.w3.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.4.w1.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.4.w2.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.4.w3.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.5.w1.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.5.w2.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.5.w3.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.6.w1.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.6.w2.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.6.w3.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.7.w1.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.7.w2.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.experts.7.w3.weight": "model-00021-of-00059.safetensors", + "model.layers.19.block_sparse_moe.gate.weight": "model-00020-of-00059.safetensors", + "model.layers.19.input_layernorm.weight": "model-00021-of-00059.safetensors", + "model.layers.19.post_attention_layernorm.weight": "model-00021-of-00059.safetensors", + "model.layers.19.self_attn.k_proj.weight": "model-00020-of-00059.safetensors", + "model.layers.19.self_attn.o_proj.weight": "model-00020-of-00059.safetensors", + "model.layers.19.self_attn.q_proj.weight": "model-00020-of-00059.safetensors", + "model.layers.19.self_attn.v_proj.weight": "model-00020-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.0.w1.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.0.w2.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.0.w3.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.1.w1.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.1.w2.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.1.w3.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.2.w1.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.2.w2.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.2.w3.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.3.w1.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.3.w2.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.3.w3.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.4.w1.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.4.w2.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.4.w3.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.5.w1.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.5.w2.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.5.w3.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.6.w1.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.6.w2.weight": "model-00003-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.6.w3.weight": "model-00004-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.7.w1.weight": "model-00004-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.7.w2.weight": "model-00004-of-00059.safetensors", + "model.layers.2.block_sparse_moe.experts.7.w3.weight": "model-00004-of-00059.safetensors", + "model.layers.2.block_sparse_moe.gate.weight": "model-00003-of-00059.safetensors", + "model.layers.2.input_layernorm.weight": "model-00004-of-00059.safetensors", + "model.layers.2.post_attention_layernorm.weight": "model-00004-of-00059.safetensors", + "model.layers.2.self_attn.k_proj.weight": "model-00003-of-00059.safetensors", + "model.layers.2.self_attn.o_proj.weight": "model-00003-of-00059.safetensors", + "model.layers.2.self_attn.q_proj.weight": "model-00003-of-00059.safetensors", + "model.layers.2.self_attn.v_proj.weight": "model-00003-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.0.w1.weight": "model-00021-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.0.w2.weight": "model-00021-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.0.w3.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.1.w1.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.1.w2.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.1.w3.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.2.w1.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.2.w2.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.2.w3.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.3.w1.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.3.w2.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.3.w3.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.4.w1.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.4.w2.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.4.w3.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.5.w1.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.5.w2.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.5.w3.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.6.w1.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.6.w2.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.6.w3.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.7.w1.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.7.w2.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.experts.7.w3.weight": "model-00022-of-00059.safetensors", + "model.layers.20.block_sparse_moe.gate.weight": "model-00021-of-00059.safetensors", + "model.layers.20.input_layernorm.weight": "model-00022-of-00059.safetensors", + "model.layers.20.post_attention_layernorm.weight": "model-00022-of-00059.safetensors", + "model.layers.20.self_attn.k_proj.weight": "model-00021-of-00059.safetensors", + "model.layers.20.self_attn.o_proj.weight": "model-00021-of-00059.safetensors", + "model.layers.20.self_attn.q_proj.weight": "model-00021-of-00059.safetensors", + "model.layers.20.self_attn.v_proj.weight": "model-00021-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.0.w1.weight": "model-00022-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.0.w2.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.0.w3.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.1.w1.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.1.w2.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.1.w3.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.2.w1.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.2.w2.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.2.w3.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.3.w1.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.3.w2.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.3.w3.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.4.w1.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.4.w2.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.4.w3.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.5.w1.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.5.w2.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.5.w3.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.6.w1.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.6.w2.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.6.w3.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.7.w1.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.7.w2.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.experts.7.w3.weight": "model-00023-of-00059.safetensors", + "model.layers.21.block_sparse_moe.gate.weight": "model-00022-of-00059.safetensors", + "model.layers.21.input_layernorm.weight": "model-00023-of-00059.safetensors", + "model.layers.21.post_attention_layernorm.weight": "model-00023-of-00059.safetensors", + "model.layers.21.self_attn.k_proj.weight": "model-00022-of-00059.safetensors", + "model.layers.21.self_attn.o_proj.weight": "model-00022-of-00059.safetensors", + "model.layers.21.self_attn.q_proj.weight": "model-00022-of-00059.safetensors", + "model.layers.21.self_attn.v_proj.weight": "model-00022-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.0.w1.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.0.w2.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.0.w3.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.1.w1.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.1.w2.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.1.w3.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.2.w1.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.2.w2.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.2.w3.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.3.w1.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.3.w2.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.3.w3.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.4.w1.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.4.w2.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.4.w3.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.5.w1.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.5.w2.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.5.w3.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.6.w1.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.6.w2.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.6.w3.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.7.w1.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.7.w2.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.experts.7.w3.weight": "model-00024-of-00059.safetensors", + "model.layers.22.block_sparse_moe.gate.weight": "model-00023-of-00059.safetensors", + "model.layers.22.input_layernorm.weight": "model-00024-of-00059.safetensors", + "model.layers.22.post_attention_layernorm.weight": "model-00024-of-00059.safetensors", + "model.layers.22.self_attn.k_proj.weight": "model-00023-of-00059.safetensors", + "model.layers.22.self_attn.o_proj.weight": "model-00023-of-00059.safetensors", + "model.layers.22.self_attn.q_proj.weight": "model-00023-of-00059.safetensors", + "model.layers.22.self_attn.v_proj.weight": "model-00023-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.0.w1.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.0.w2.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.0.w3.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.1.w1.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.1.w2.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.1.w3.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.2.w1.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.2.w2.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.2.w3.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.3.w1.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.3.w2.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.3.w3.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.4.w1.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.4.w2.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.4.w3.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.5.w1.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.5.w2.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.5.w3.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.6.w1.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.6.w2.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.6.w3.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.7.w1.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.7.w2.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.experts.7.w3.weight": "model-00025-of-00059.safetensors", + "model.layers.23.block_sparse_moe.gate.weight": "model-00025-of-00059.safetensors", + "model.layers.23.input_layernorm.weight": "model-00025-of-00059.safetensors", + "model.layers.23.post_attention_layernorm.weight": "model-00025-of-00059.safetensors", + "model.layers.23.self_attn.k_proj.weight": "model-00024-of-00059.safetensors", + "model.layers.23.self_attn.o_proj.weight": "model-00025-of-00059.safetensors", + "model.layers.23.self_attn.q_proj.weight": "model-00024-of-00059.safetensors", + "model.layers.23.self_attn.v_proj.weight": "model-00024-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.0.w1.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.0.w2.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.0.w3.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.1.w1.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.1.w2.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.1.w3.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.2.w1.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.2.w2.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.2.w3.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.3.w1.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.3.w2.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.3.w3.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.4.w1.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.4.w2.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.4.w3.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.5.w1.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.5.w2.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.5.w3.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.6.w1.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.6.w2.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.6.w3.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.7.w1.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.7.w2.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.experts.7.w3.weight": "model-00026-of-00059.safetensors", + "model.layers.24.block_sparse_moe.gate.weight": "model-00026-of-00059.safetensors", + "model.layers.24.input_layernorm.weight": "model-00026-of-00059.safetensors", + "model.layers.24.post_attention_layernorm.weight": "model-00026-of-00059.safetensors", + "model.layers.24.self_attn.k_proj.weight": "model-00025-of-00059.safetensors", + "model.layers.24.self_attn.o_proj.weight": "model-00026-of-00059.safetensors", + "model.layers.24.self_attn.q_proj.weight": "model-00025-of-00059.safetensors", + "model.layers.24.self_attn.v_proj.weight": "model-00026-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.0.w1.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.0.w2.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.0.w3.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.1.w1.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.1.w2.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.1.w3.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.2.w1.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.2.w2.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.2.w3.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.3.w1.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.3.w2.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.3.w3.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.4.w1.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.4.w2.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.4.w3.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.5.w1.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.5.w2.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.5.w3.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.6.w1.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.6.w2.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.6.w3.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.7.w1.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.7.w2.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.experts.7.w3.weight": "model-00027-of-00059.safetensors", + "model.layers.25.block_sparse_moe.gate.weight": "model-00027-of-00059.safetensors", + "model.layers.25.input_layernorm.weight": "model-00027-of-00059.safetensors", + "model.layers.25.post_attention_layernorm.weight": "model-00027-of-00059.safetensors", + "model.layers.25.self_attn.k_proj.weight": "model-00027-of-00059.safetensors", + "model.layers.25.self_attn.o_proj.weight": "model-00027-of-00059.safetensors", + "model.layers.25.self_attn.q_proj.weight": "model-00026-of-00059.safetensors", + "model.layers.25.self_attn.v_proj.weight": "model-00027-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.0.w1.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.0.w2.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.0.w3.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.1.w1.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.1.w2.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.1.w3.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.2.w1.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.2.w2.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.2.w3.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.3.w1.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.3.w2.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.3.w3.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.4.w1.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.4.w2.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.4.w3.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.5.w1.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.5.w2.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.5.w3.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.6.w1.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.6.w2.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.6.w3.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.7.w1.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.7.w2.weight": "model-00028-of-00059.safetensors", + "model.layers.26.block_sparse_moe.experts.7.w3.weight": "model-00029-of-00059.safetensors", + "model.layers.26.block_sparse_moe.gate.weight": "model-00028-of-00059.safetensors", + "model.layers.26.input_layernorm.weight": "model-00029-of-00059.safetensors", + "model.layers.26.post_attention_layernorm.weight": "model-00029-of-00059.safetensors", + "model.layers.26.self_attn.k_proj.weight": "model-00028-of-00059.safetensors", + "model.layers.26.self_attn.o_proj.weight": "model-00028-of-00059.safetensors", + "model.layers.26.self_attn.q_proj.weight": "model-00028-of-00059.safetensors", + "model.layers.26.self_attn.v_proj.weight": "model-00028-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.0.w1.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.0.w2.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.0.w3.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.1.w1.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.1.w2.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.1.w3.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.2.w1.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.2.w2.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.2.w3.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.3.w1.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.3.w2.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.3.w3.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.4.w1.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.4.w2.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.4.w3.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.5.w1.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.5.w2.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.5.w3.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.6.w1.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.6.w2.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.6.w3.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.7.w1.weight": "model-00029-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.7.w2.weight": "model-00030-of-00059.safetensors", + "model.layers.27.block_sparse_moe.experts.7.w3.weight": "model-00030-of-00059.safetensors", + "model.layers.27.block_sparse_moe.gate.weight": "model-00029-of-00059.safetensors", + "model.layers.27.input_layernorm.weight": "model-00030-of-00059.safetensors", + "model.layers.27.post_attention_layernorm.weight": "model-00030-of-00059.safetensors", + "model.layers.27.self_attn.k_proj.weight": "model-00029-of-00059.safetensors", + "model.layers.27.self_attn.o_proj.weight": "model-00029-of-00059.safetensors", + "model.layers.27.self_attn.q_proj.weight": "model-00029-of-00059.safetensors", + "model.layers.27.self_attn.v_proj.weight": "model-00029-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.0.w1.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.0.w2.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.0.w3.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.1.w1.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.1.w2.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.1.w3.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.2.w1.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.2.w2.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.2.w3.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.3.w1.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.3.w2.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.3.w3.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.4.w1.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.4.w2.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.4.w3.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.5.w1.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.5.w2.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.5.w3.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.6.w1.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.6.w2.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.6.w3.weight": "model-00030-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.7.w1.weight": "model-00031-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.7.w2.weight": "model-00031-of-00059.safetensors", + "model.layers.28.block_sparse_moe.experts.7.w3.weight": "model-00031-of-00059.safetensors", + "model.layers.28.block_sparse_moe.gate.weight": "model-00030-of-00059.safetensors", + "model.layers.28.input_layernorm.weight": "model-00031-of-00059.safetensors", + "model.layers.28.post_attention_layernorm.weight": "model-00031-of-00059.safetensors", + "model.layers.28.self_attn.k_proj.weight": "model-00030-of-00059.safetensors", + "model.layers.28.self_attn.o_proj.weight": "model-00030-of-00059.safetensors", + "model.layers.28.self_attn.q_proj.weight": "model-00030-of-00059.safetensors", + "model.layers.28.self_attn.v_proj.weight": "model-00030-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.0.w1.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.0.w2.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.0.w3.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.1.w1.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.1.w2.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.1.w3.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.2.w1.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.2.w2.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.2.w3.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.3.w1.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.3.w2.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.3.w3.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.4.w1.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.4.w2.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.4.w3.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.5.w1.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.5.w2.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.5.w3.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.6.w1.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.6.w2.weight": "model-00031-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.6.w3.weight": "model-00032-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.7.w1.weight": "model-00032-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.7.w2.weight": "model-00032-of-00059.safetensors", + "model.layers.29.block_sparse_moe.experts.7.w3.weight": "model-00032-of-00059.safetensors", + "model.layers.29.block_sparse_moe.gate.weight": "model-00031-of-00059.safetensors", + "model.layers.29.input_layernorm.weight": "model-00032-of-00059.safetensors", + "model.layers.29.post_attention_layernorm.weight": "model-00032-of-00059.safetensors", + "model.layers.29.self_attn.k_proj.weight": "model-00031-of-00059.safetensors", + "model.layers.29.self_attn.o_proj.weight": "model-00031-of-00059.safetensors", + "model.layers.29.self_attn.q_proj.weight": "model-00031-of-00059.safetensors", + "model.layers.29.self_attn.v_proj.weight": "model-00031-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.0.w1.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.0.w2.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.0.w3.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.1.w1.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.1.w2.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.1.w3.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.2.w1.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.2.w2.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.2.w3.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.3.w1.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.3.w2.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.3.w3.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.4.w1.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.4.w2.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.4.w3.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.5.w1.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.5.w2.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.5.w3.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.6.w1.weight": "model-00004-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.6.w2.weight": "model-00005-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.6.w3.weight": "model-00005-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.7.w1.weight": "model-00005-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.7.w2.weight": "model-00005-of-00059.safetensors", + "model.layers.3.block_sparse_moe.experts.7.w3.weight": "model-00005-of-00059.safetensors", + "model.layers.3.block_sparse_moe.gate.weight": "model-00004-of-00059.safetensors", + "model.layers.3.input_layernorm.weight": "model-00005-of-00059.safetensors", + "model.layers.3.post_attention_layernorm.weight": "model-00005-of-00059.safetensors", + "model.layers.3.self_attn.k_proj.weight": "model-00004-of-00059.safetensors", + "model.layers.3.self_attn.o_proj.weight": "model-00004-of-00059.safetensors", + "model.layers.3.self_attn.q_proj.weight": "model-00004-of-00059.safetensors", + "model.layers.3.self_attn.v_proj.weight": "model-00004-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.0.w1.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.0.w2.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.0.w3.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.1.w1.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.1.w2.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.1.w3.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.2.w1.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.2.w2.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.2.w3.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.3.w1.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.3.w2.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.3.w3.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.4.w1.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.4.w2.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.4.w3.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.5.w1.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.5.w2.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.5.w3.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.6.w1.weight": "model-00032-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.6.w2.weight": "model-00033-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.6.w3.weight": "model-00033-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.7.w1.weight": "model-00033-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.7.w2.weight": "model-00033-of-00059.safetensors", + "model.layers.30.block_sparse_moe.experts.7.w3.weight": "model-00033-of-00059.safetensors", + "model.layers.30.block_sparse_moe.gate.weight": "model-00032-of-00059.safetensors", + "model.layers.30.input_layernorm.weight": "model-00033-of-00059.safetensors", + "model.layers.30.post_attention_layernorm.weight": "model-00033-of-00059.safetensors", + "model.layers.30.self_attn.k_proj.weight": "model-00032-of-00059.safetensors", + "model.layers.30.self_attn.o_proj.weight": "model-00032-of-00059.safetensors", + "model.layers.30.self_attn.q_proj.weight": "model-00032-of-00059.safetensors", + "model.layers.30.self_attn.v_proj.weight": "model-00032-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.0.w1.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.0.w2.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.0.w3.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.1.w1.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.1.w2.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.1.w3.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.2.w1.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.2.w2.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.2.w3.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.3.w1.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.3.w2.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.3.w3.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.4.w1.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.4.w2.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.4.w3.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.5.w1.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.5.w2.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.5.w3.weight": "model-00033-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.6.w1.weight": "model-00034-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.6.w2.weight": "model-00034-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.6.w3.weight": "model-00034-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.7.w1.weight": "model-00034-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.7.w2.weight": "model-00034-of-00059.safetensors", + "model.layers.31.block_sparse_moe.experts.7.w3.weight": "model-00034-of-00059.safetensors", + "model.layers.31.block_sparse_moe.gate.weight": "model-00033-of-00059.safetensors", + "model.layers.31.input_layernorm.weight": "model-00034-of-00059.safetensors", + "model.layers.31.post_attention_layernorm.weight": "model-00034-of-00059.safetensors", + "model.layers.31.self_attn.k_proj.weight": "model-00033-of-00059.safetensors", + "model.layers.31.self_attn.o_proj.weight": "model-00033-of-00059.safetensors", + "model.layers.31.self_attn.q_proj.weight": "model-00033-of-00059.safetensors", + "model.layers.31.self_attn.v_proj.weight": "model-00033-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.0.w1.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.0.w2.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.0.w3.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.1.w1.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.1.w2.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.1.w3.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.2.w1.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.2.w2.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.2.w3.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.3.w1.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.3.w2.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.3.w3.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.4.w1.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.4.w2.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.4.w3.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.5.w1.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.5.w2.weight": "model-00034-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.5.w3.weight": "model-00035-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.6.w1.weight": "model-00035-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.6.w2.weight": "model-00035-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.6.w3.weight": "model-00035-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.7.w1.weight": "model-00035-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.7.w2.weight": "model-00035-of-00059.safetensors", + "model.layers.32.block_sparse_moe.experts.7.w3.weight": "model-00035-of-00059.safetensors", + "model.layers.32.block_sparse_moe.gate.weight": "model-00034-of-00059.safetensors", + "model.layers.32.input_layernorm.weight": "model-00035-of-00059.safetensors", + "model.layers.32.post_attention_layernorm.weight": "model-00035-of-00059.safetensors", + "model.layers.32.self_attn.k_proj.weight": "model-00034-of-00059.safetensors", + "model.layers.32.self_attn.o_proj.weight": "model-00034-of-00059.safetensors", + "model.layers.32.self_attn.q_proj.weight": "model-00034-of-00059.safetensors", + "model.layers.32.self_attn.v_proj.weight": "model-00034-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.0.w1.weight": "model-00035-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.0.w2.weight": "model-00035-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.0.w3.weight": "model-00035-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.1.w1.weight": "model-00035-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.1.w2.weight": "model-00035-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.1.w3.weight": "model-00035-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.2.w1.weight": "model-00035-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.2.w2.weight": "model-00035-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.2.w3.weight": "model-00035-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.3.w1.weight": "model-00035-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.3.w2.weight": "model-00035-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.3.w3.weight": "model-00035-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.4.w1.weight": "model-00035-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.4.w2.weight": "model-00035-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.4.w3.weight": "model-00035-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.5.w1.weight": "model-00035-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.5.w2.weight": "model-00036-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.5.w3.weight": "model-00036-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.6.w1.weight": "model-00036-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.6.w2.weight": "model-00036-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.6.w3.weight": "model-00036-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.7.w1.weight": "model-00036-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.7.w2.weight": "model-00036-of-00059.safetensors", + "model.layers.33.block_sparse_moe.experts.7.w3.weight": "model-00036-of-00059.safetensors", + "model.layers.33.block_sparse_moe.gate.weight": "model-00035-of-00059.safetensors", + "model.layers.33.input_layernorm.weight": "model-00036-of-00059.safetensors", + "model.layers.33.post_attention_layernorm.weight": "model-00036-of-00059.safetensors", + "model.layers.33.self_attn.k_proj.weight": "model-00035-of-00059.safetensors", + "model.layers.33.self_attn.o_proj.weight": "model-00035-of-00059.safetensors", + "model.layers.33.self_attn.q_proj.weight": "model-00035-of-00059.safetensors", + "model.layers.33.self_attn.v_proj.weight": "model-00035-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.0.w1.weight": "model-00036-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.0.w2.weight": "model-00036-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.0.w3.weight": "model-00036-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.1.w1.weight": "model-00036-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.1.w2.weight": "model-00036-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.1.w3.weight": "model-00036-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.2.w1.weight": "model-00036-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.2.w2.weight": "model-00036-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.2.w3.weight": "model-00036-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.3.w1.weight": "model-00036-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.3.w2.weight": "model-00036-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.3.w3.weight": "model-00036-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.4.w1.weight": "model-00036-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.4.w2.weight": "model-00036-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.4.w3.weight": "model-00036-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.5.w1.weight": "model-00037-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.5.w2.weight": "model-00037-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.5.w3.weight": "model-00037-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.6.w1.weight": "model-00037-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.6.w2.weight": "model-00037-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.6.w3.weight": "model-00037-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.7.w1.weight": "model-00037-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.7.w2.weight": "model-00037-of-00059.safetensors", + "model.layers.34.block_sparse_moe.experts.7.w3.weight": "model-00037-of-00059.safetensors", + "model.layers.34.block_sparse_moe.gate.weight": "model-00036-of-00059.safetensors", + "model.layers.34.input_layernorm.weight": "model-00037-of-00059.safetensors", + "model.layers.34.post_attention_layernorm.weight": "model-00037-of-00059.safetensors", + "model.layers.34.self_attn.k_proj.weight": "model-00036-of-00059.safetensors", + "model.layers.34.self_attn.o_proj.weight": "model-00036-of-00059.safetensors", + "model.layers.34.self_attn.q_proj.weight": "model-00036-of-00059.safetensors", + "model.layers.34.self_attn.v_proj.weight": "model-00036-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.0.w1.weight": "model-00037-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.0.w2.weight": "model-00037-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.0.w3.weight": "model-00037-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.1.w1.weight": "model-00037-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.1.w2.weight": "model-00037-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.1.w3.weight": "model-00037-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.2.w1.weight": "model-00037-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.2.w2.weight": "model-00037-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.2.w3.weight": "model-00037-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.3.w1.weight": "model-00037-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.3.w2.weight": "model-00037-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.3.w3.weight": "model-00037-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.4.w1.weight": "model-00037-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.4.w2.weight": "model-00037-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.4.w3.weight": "model-00038-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.5.w1.weight": "model-00038-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.5.w2.weight": "model-00038-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.5.w3.weight": "model-00038-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.6.w1.weight": "model-00038-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.6.w2.weight": "model-00038-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.6.w3.weight": "model-00038-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.7.w1.weight": "model-00038-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.7.w2.weight": "model-00038-of-00059.safetensors", + "model.layers.35.block_sparse_moe.experts.7.w3.weight": "model-00038-of-00059.safetensors", + "model.layers.35.block_sparse_moe.gate.weight": "model-00037-of-00059.safetensors", + "model.layers.35.input_layernorm.weight": "model-00038-of-00059.safetensors", + "model.layers.35.post_attention_layernorm.weight": "model-00038-of-00059.safetensors", + "model.layers.35.self_attn.k_proj.weight": "model-00037-of-00059.safetensors", + "model.layers.35.self_attn.o_proj.weight": "model-00037-of-00059.safetensors", + "model.layers.35.self_attn.q_proj.weight": "model-00037-of-00059.safetensors", + "model.layers.35.self_attn.v_proj.weight": "model-00037-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.0.w1.weight": "model-00038-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.0.w2.weight": "model-00038-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.0.w3.weight": "model-00038-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.1.w1.weight": "model-00038-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.1.w2.weight": "model-00038-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.1.w3.weight": "model-00038-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.2.w1.weight": "model-00038-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.2.w2.weight": "model-00038-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.2.w3.weight": "model-00038-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.3.w1.weight": "model-00038-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.3.w2.weight": "model-00038-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.3.w3.weight": "model-00038-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.4.w1.weight": "model-00038-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.4.w2.weight": "model-00039-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.4.w3.weight": "model-00039-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.5.w1.weight": "model-00039-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.5.w2.weight": "model-00039-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.5.w3.weight": "model-00039-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.6.w1.weight": "model-00039-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.6.w2.weight": "model-00039-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.6.w3.weight": "model-00039-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.7.w1.weight": "model-00039-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.7.w2.weight": "model-00039-of-00059.safetensors", + "model.layers.36.block_sparse_moe.experts.7.w3.weight": "model-00039-of-00059.safetensors", + "model.layers.36.block_sparse_moe.gate.weight": "model-00038-of-00059.safetensors", + "model.layers.36.input_layernorm.weight": "model-00039-of-00059.safetensors", + "model.layers.36.post_attention_layernorm.weight": "model-00039-of-00059.safetensors", + "model.layers.36.self_attn.k_proj.weight": "model-00038-of-00059.safetensors", + "model.layers.36.self_attn.o_proj.weight": "model-00038-of-00059.safetensors", + "model.layers.36.self_attn.q_proj.weight": "model-00038-of-00059.safetensors", + "model.layers.36.self_attn.v_proj.weight": "model-00038-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.0.w1.weight": "model-00039-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.0.w2.weight": "model-00039-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.0.w3.weight": "model-00039-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.1.w1.weight": "model-00039-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.1.w2.weight": "model-00039-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.1.w3.weight": "model-00039-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.2.w1.weight": "model-00039-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.2.w2.weight": "model-00039-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.2.w3.weight": "model-00039-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.3.w1.weight": "model-00039-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.3.w2.weight": "model-00039-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.3.w3.weight": "model-00039-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.4.w1.weight": "model-00040-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.4.w2.weight": "model-00040-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.4.w3.weight": "model-00040-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.5.w1.weight": "model-00040-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.5.w2.weight": "model-00040-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.5.w3.weight": "model-00040-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.6.w1.weight": "model-00040-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.6.w2.weight": "model-00040-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.6.w3.weight": "model-00040-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.7.w1.weight": "model-00040-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.7.w2.weight": "model-00040-of-00059.safetensors", + "model.layers.37.block_sparse_moe.experts.7.w3.weight": "model-00040-of-00059.safetensors", + "model.layers.37.block_sparse_moe.gate.weight": "model-00039-of-00059.safetensors", + "model.layers.37.input_layernorm.weight": "model-00040-of-00059.safetensors", + "model.layers.37.post_attention_layernorm.weight": "model-00040-of-00059.safetensors", + "model.layers.37.self_attn.k_proj.weight": "model-00039-of-00059.safetensors", + "model.layers.37.self_attn.o_proj.weight": "model-00039-of-00059.safetensors", + "model.layers.37.self_attn.q_proj.weight": "model-00039-of-00059.safetensors", + "model.layers.37.self_attn.v_proj.weight": "model-00039-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.0.w1.weight": "model-00040-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.0.w2.weight": "model-00040-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.0.w3.weight": "model-00040-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.1.w1.weight": "model-00040-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.1.w2.weight": "model-00040-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.1.w3.weight": "model-00040-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.2.w1.weight": "model-00040-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.2.w2.weight": "model-00040-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.2.w3.weight": "model-00040-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.3.w1.weight": "model-00040-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.3.w2.weight": "model-00040-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.3.w3.weight": "model-00041-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.4.w1.weight": "model-00041-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.4.w2.weight": "model-00041-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.4.w3.weight": "model-00041-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.5.w1.weight": "model-00041-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.5.w2.weight": "model-00041-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.5.w3.weight": "model-00041-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.6.w1.weight": "model-00041-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.6.w2.weight": "model-00041-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.6.w3.weight": "model-00041-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.7.w1.weight": "model-00041-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.7.w2.weight": "model-00041-of-00059.safetensors", + "model.layers.38.block_sparse_moe.experts.7.w3.weight": "model-00041-of-00059.safetensors", + "model.layers.38.block_sparse_moe.gate.weight": "model-00040-of-00059.safetensors", + "model.layers.38.input_layernorm.weight": "model-00041-of-00059.safetensors", + "model.layers.38.post_attention_layernorm.weight": "model-00041-of-00059.safetensors", + "model.layers.38.self_attn.k_proj.weight": "model-00040-of-00059.safetensors", + "model.layers.38.self_attn.o_proj.weight": "model-00040-of-00059.safetensors", + "model.layers.38.self_attn.q_proj.weight": "model-00040-of-00059.safetensors", + "model.layers.38.self_attn.v_proj.weight": "model-00040-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.0.w1.weight": "model-00041-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.0.w2.weight": "model-00041-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.0.w3.weight": "model-00041-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.1.w1.weight": "model-00041-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.1.w2.weight": "model-00041-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.1.w3.weight": "model-00041-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.2.w1.weight": "model-00041-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.2.w2.weight": "model-00041-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.2.w3.weight": "model-00041-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.3.w1.weight": "model-00041-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.3.w2.weight": "model-00042-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.3.w3.weight": "model-00042-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.4.w1.weight": "model-00042-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.4.w2.weight": "model-00042-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.4.w3.weight": "model-00042-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.5.w1.weight": "model-00042-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.5.w2.weight": "model-00042-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.5.w3.weight": "model-00042-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.6.w1.weight": "model-00042-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.6.w2.weight": "model-00042-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.6.w3.weight": "model-00042-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.7.w1.weight": "model-00042-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.7.w2.weight": "model-00042-of-00059.safetensors", + "model.layers.39.block_sparse_moe.experts.7.w3.weight": "model-00042-of-00059.safetensors", + "model.layers.39.block_sparse_moe.gate.weight": "model-00041-of-00059.safetensors", + "model.layers.39.input_layernorm.weight": "model-00042-of-00059.safetensors", + "model.layers.39.post_attention_layernorm.weight": "model-00042-of-00059.safetensors", + "model.layers.39.self_attn.k_proj.weight": "model-00041-of-00059.safetensors", + "model.layers.39.self_attn.o_proj.weight": "model-00041-of-00059.safetensors", + "model.layers.39.self_attn.q_proj.weight": "model-00041-of-00059.safetensors", + "model.layers.39.self_attn.v_proj.weight": "model-00041-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.0.w1.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.0.w2.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.0.w3.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.1.w1.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.1.w2.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.1.w3.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.2.w1.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.2.w2.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.2.w3.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.3.w1.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.3.w2.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.3.w3.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.4.w1.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.4.w2.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.4.w3.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.5.w1.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.5.w2.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.5.w3.weight": "model-00005-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.6.w1.weight": "model-00006-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.6.w2.weight": "model-00006-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.6.w3.weight": "model-00006-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.7.w1.weight": "model-00006-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.7.w2.weight": "model-00006-of-00059.safetensors", + "model.layers.4.block_sparse_moe.experts.7.w3.weight": "model-00006-of-00059.safetensors", + "model.layers.4.block_sparse_moe.gate.weight": "model-00005-of-00059.safetensors", + "model.layers.4.input_layernorm.weight": "model-00006-of-00059.safetensors", + "model.layers.4.post_attention_layernorm.weight": "model-00006-of-00059.safetensors", + "model.layers.4.self_attn.k_proj.weight": "model-00005-of-00059.safetensors", + "model.layers.4.self_attn.o_proj.weight": "model-00005-of-00059.safetensors", + "model.layers.4.self_attn.q_proj.weight": "model-00005-of-00059.safetensors", + "model.layers.4.self_attn.v_proj.weight": "model-00005-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.0.w1.weight": "model-00042-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.0.w2.weight": "model-00042-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.0.w3.weight": "model-00042-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.1.w1.weight": "model-00042-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.1.w2.weight": "model-00042-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.1.w3.weight": "model-00042-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.2.w1.weight": "model-00042-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.2.w2.weight": "model-00042-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.2.w3.weight": "model-00042-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.3.w1.weight": "model-00043-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.3.w2.weight": "model-00043-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.3.w3.weight": "model-00043-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.4.w1.weight": "model-00043-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.4.w2.weight": "model-00043-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.4.w3.weight": "model-00043-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.5.w1.weight": "model-00043-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.5.w2.weight": "model-00043-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.5.w3.weight": "model-00043-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.6.w1.weight": "model-00043-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.6.w2.weight": "model-00043-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.6.w3.weight": "model-00043-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.7.w1.weight": "model-00043-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.7.w2.weight": "model-00043-of-00059.safetensors", + "model.layers.40.block_sparse_moe.experts.7.w3.weight": "model-00043-of-00059.safetensors", + "model.layers.40.block_sparse_moe.gate.weight": "model-00042-of-00059.safetensors", + "model.layers.40.input_layernorm.weight": "model-00043-of-00059.safetensors", + "model.layers.40.post_attention_layernorm.weight": "model-00043-of-00059.safetensors", + "model.layers.40.self_attn.k_proj.weight": "model-00042-of-00059.safetensors", + "model.layers.40.self_attn.o_proj.weight": "model-00042-of-00059.safetensors", + "model.layers.40.self_attn.q_proj.weight": "model-00042-of-00059.safetensors", + "model.layers.40.self_attn.v_proj.weight": "model-00042-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.0.w1.weight": "model-00043-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.0.w2.weight": "model-00043-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.0.w3.weight": "model-00043-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.1.w1.weight": "model-00043-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.1.w2.weight": "model-00043-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.1.w3.weight": "model-00043-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.2.w1.weight": "model-00043-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.2.w2.weight": "model-00043-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.2.w3.weight": "model-00044-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.3.w1.weight": "model-00044-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.3.w2.weight": "model-00044-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.3.w3.weight": "model-00044-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.4.w1.weight": "model-00044-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.4.w2.weight": "model-00044-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.4.w3.weight": "model-00044-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.5.w1.weight": "model-00044-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.5.w2.weight": "model-00044-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.5.w3.weight": "model-00044-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.6.w1.weight": "model-00044-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.6.w2.weight": "model-00044-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.6.w3.weight": "model-00044-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.7.w1.weight": "model-00044-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.7.w2.weight": "model-00044-of-00059.safetensors", + "model.layers.41.block_sparse_moe.experts.7.w3.weight": "model-00044-of-00059.safetensors", + "model.layers.41.block_sparse_moe.gate.weight": "model-00043-of-00059.safetensors", + "model.layers.41.input_layernorm.weight": "model-00044-of-00059.safetensors", + "model.layers.41.post_attention_layernorm.weight": "model-00044-of-00059.safetensors", + "model.layers.41.self_attn.k_proj.weight": "model-00043-of-00059.safetensors", + "model.layers.41.self_attn.o_proj.weight": "model-00043-of-00059.safetensors", + "model.layers.41.self_attn.q_proj.weight": "model-00043-of-00059.safetensors", + "model.layers.41.self_attn.v_proj.weight": "model-00043-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.0.w1.weight": "model-00044-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.0.w2.weight": "model-00044-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.0.w3.weight": "model-00044-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.1.w1.weight": "model-00044-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.1.w2.weight": "model-00044-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.1.w3.weight": "model-00044-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.2.w1.weight": "model-00044-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.2.w2.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.2.w3.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.3.w1.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.3.w2.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.3.w3.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.4.w1.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.4.w2.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.4.w3.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.5.w1.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.5.w2.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.5.w3.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.6.w1.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.6.w2.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.6.w3.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.7.w1.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.7.w2.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.experts.7.w3.weight": "model-00045-of-00059.safetensors", + "model.layers.42.block_sparse_moe.gate.weight": "model-00044-of-00059.safetensors", + "model.layers.42.input_layernorm.weight": "model-00045-of-00059.safetensors", + "model.layers.42.post_attention_layernorm.weight": "model-00045-of-00059.safetensors", + "model.layers.42.self_attn.k_proj.weight": "model-00044-of-00059.safetensors", + "model.layers.42.self_attn.o_proj.weight": "model-00044-of-00059.safetensors", + "model.layers.42.self_attn.q_proj.weight": "model-00044-of-00059.safetensors", + "model.layers.42.self_attn.v_proj.weight": "model-00044-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.0.w1.weight": "model-00045-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.0.w2.weight": "model-00045-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.0.w3.weight": "model-00045-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.1.w1.weight": "model-00045-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.1.w2.weight": "model-00045-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.1.w3.weight": "model-00045-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.2.w1.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.2.w2.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.2.w3.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.3.w1.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.3.w2.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.3.w3.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.4.w1.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.4.w2.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.4.w3.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.5.w1.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.5.w2.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.5.w3.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.6.w1.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.6.w2.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.6.w3.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.7.w1.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.7.w2.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.experts.7.w3.weight": "model-00046-of-00059.safetensors", + "model.layers.43.block_sparse_moe.gate.weight": "model-00045-of-00059.safetensors", + "model.layers.43.input_layernorm.weight": "model-00046-of-00059.safetensors", + "model.layers.43.post_attention_layernorm.weight": "model-00046-of-00059.safetensors", + "model.layers.43.self_attn.k_proj.weight": "model-00045-of-00059.safetensors", + "model.layers.43.self_attn.o_proj.weight": "model-00045-of-00059.safetensors", + "model.layers.43.self_attn.q_proj.weight": "model-00045-of-00059.safetensors", + "model.layers.43.self_attn.v_proj.weight": "model-00045-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.0.w1.weight": "model-00046-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.0.w2.weight": "model-00046-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.0.w3.weight": "model-00046-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.1.w1.weight": "model-00046-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.1.w2.weight": "model-00046-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.1.w3.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.2.w1.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.2.w2.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.2.w3.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.3.w1.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.3.w2.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.3.w3.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.4.w1.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.4.w2.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.4.w3.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.5.w1.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.5.w2.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.5.w3.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.6.w1.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.6.w2.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.6.w3.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.7.w1.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.7.w2.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.experts.7.w3.weight": "model-00047-of-00059.safetensors", + "model.layers.44.block_sparse_moe.gate.weight": "model-00046-of-00059.safetensors", + "model.layers.44.input_layernorm.weight": "model-00047-of-00059.safetensors", + "model.layers.44.post_attention_layernorm.weight": "model-00047-of-00059.safetensors", + "model.layers.44.self_attn.k_proj.weight": "model-00046-of-00059.safetensors", + "model.layers.44.self_attn.o_proj.weight": "model-00046-of-00059.safetensors", + "model.layers.44.self_attn.q_proj.weight": "model-00046-of-00059.safetensors", + "model.layers.44.self_attn.v_proj.weight": "model-00046-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.0.w1.weight": "model-00047-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.0.w2.weight": "model-00047-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.0.w3.weight": "model-00047-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.1.w1.weight": "model-00047-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.1.w2.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.1.w3.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.2.w1.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.2.w2.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.2.w3.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.3.w1.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.3.w2.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.3.w3.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.4.w1.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.4.w2.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.4.w3.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.5.w1.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.5.w2.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.5.w3.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.6.w1.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.6.w2.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.6.w3.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.7.w1.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.7.w2.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.experts.7.w3.weight": "model-00048-of-00059.safetensors", + "model.layers.45.block_sparse_moe.gate.weight": "model-00047-of-00059.safetensors", + "model.layers.45.input_layernorm.weight": "model-00048-of-00059.safetensors", + "model.layers.45.post_attention_layernorm.weight": "model-00048-of-00059.safetensors", + "model.layers.45.self_attn.k_proj.weight": "model-00047-of-00059.safetensors", + "model.layers.45.self_attn.o_proj.weight": "model-00047-of-00059.safetensors", + "model.layers.45.self_attn.q_proj.weight": "model-00047-of-00059.safetensors", + "model.layers.45.self_attn.v_proj.weight": "model-00047-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.0.w1.weight": "model-00048-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.0.w2.weight": "model-00048-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.0.w3.weight": "model-00048-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.1.w1.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.1.w2.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.1.w3.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.2.w1.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.2.w2.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.2.w3.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.3.w1.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.3.w2.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.3.w3.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.4.w1.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.4.w2.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.4.w3.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.5.w1.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.5.w2.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.5.w3.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.6.w1.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.6.w2.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.6.w3.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.7.w1.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.7.w2.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.experts.7.w3.weight": "model-00049-of-00059.safetensors", + "model.layers.46.block_sparse_moe.gate.weight": "model-00048-of-00059.safetensors", + "model.layers.46.input_layernorm.weight": "model-00049-of-00059.safetensors", + "model.layers.46.post_attention_layernorm.weight": "model-00049-of-00059.safetensors", + "model.layers.46.self_attn.k_proj.weight": "model-00048-of-00059.safetensors", + "model.layers.46.self_attn.o_proj.weight": "model-00048-of-00059.safetensors", + "model.layers.46.self_attn.q_proj.weight": "model-00048-of-00059.safetensors", + "model.layers.46.self_attn.v_proj.weight": "model-00048-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.0.w1.weight": "model-00049-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.0.w2.weight": "model-00049-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.0.w3.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.1.w1.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.1.w2.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.1.w3.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.2.w1.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.2.w2.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.2.w3.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.3.w1.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.3.w2.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.3.w3.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.4.w1.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.4.w2.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.4.w3.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.5.w1.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.5.w2.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.5.w3.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.6.w1.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.6.w2.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.6.w3.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.7.w1.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.7.w2.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.experts.7.w3.weight": "model-00050-of-00059.safetensors", + "model.layers.47.block_sparse_moe.gate.weight": "model-00049-of-00059.safetensors", + "model.layers.47.input_layernorm.weight": "model-00050-of-00059.safetensors", + "model.layers.47.post_attention_layernorm.weight": "model-00050-of-00059.safetensors", + "model.layers.47.self_attn.k_proj.weight": "model-00049-of-00059.safetensors", + "model.layers.47.self_attn.o_proj.weight": "model-00049-of-00059.safetensors", + "model.layers.47.self_attn.q_proj.weight": "model-00049-of-00059.safetensors", + "model.layers.47.self_attn.v_proj.weight": "model-00049-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.0.w1.weight": "model-00050-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.0.w2.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.0.w3.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.1.w1.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.1.w2.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.1.w3.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.2.w1.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.2.w2.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.2.w3.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.3.w1.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.3.w2.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.3.w3.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.4.w1.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.4.w2.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.4.w3.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.5.w1.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.5.w2.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.5.w3.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.6.w1.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.6.w2.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.6.w3.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.7.w1.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.7.w2.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.experts.7.w3.weight": "model-00051-of-00059.safetensors", + "model.layers.48.block_sparse_moe.gate.weight": "model-00050-of-00059.safetensors", + "model.layers.48.input_layernorm.weight": "model-00051-of-00059.safetensors", + "model.layers.48.post_attention_layernorm.weight": "model-00051-of-00059.safetensors", + "model.layers.48.self_attn.k_proj.weight": "model-00050-of-00059.safetensors", + "model.layers.48.self_attn.o_proj.weight": "model-00050-of-00059.safetensors", + "model.layers.48.self_attn.q_proj.weight": "model-00050-of-00059.safetensors", + "model.layers.48.self_attn.v_proj.weight": "model-00050-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.0.w1.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.0.w2.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.0.w3.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.1.w1.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.1.w2.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.1.w3.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.2.w1.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.2.w2.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.2.w3.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.3.w1.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.3.w2.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.3.w3.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.4.w1.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.4.w2.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.4.w3.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.5.w1.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.5.w2.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.5.w3.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.6.w1.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.6.w2.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.6.w3.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.7.w1.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.7.w2.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.experts.7.w3.weight": "model-00052-of-00059.safetensors", + "model.layers.49.block_sparse_moe.gate.weight": "model-00051-of-00059.safetensors", + "model.layers.49.input_layernorm.weight": "model-00052-of-00059.safetensors", + "model.layers.49.post_attention_layernorm.weight": "model-00052-of-00059.safetensors", + "model.layers.49.self_attn.k_proj.weight": "model-00051-of-00059.safetensors", + "model.layers.49.self_attn.o_proj.weight": "model-00051-of-00059.safetensors", + "model.layers.49.self_attn.q_proj.weight": "model-00051-of-00059.safetensors", + "model.layers.49.self_attn.v_proj.weight": "model-00051-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.0.w1.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.0.w2.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.0.w3.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.1.w1.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.1.w2.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.1.w3.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.2.w1.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.2.w2.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.2.w3.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.3.w1.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.3.w2.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.3.w3.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.4.w1.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.4.w2.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.4.w3.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.5.w1.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.5.w2.weight": "model-00006-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.5.w3.weight": "model-00007-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.6.w1.weight": "model-00007-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.6.w2.weight": "model-00007-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.6.w3.weight": "model-00007-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.7.w1.weight": "model-00007-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.7.w2.weight": "model-00007-of-00059.safetensors", + "model.layers.5.block_sparse_moe.experts.7.w3.weight": "model-00007-of-00059.safetensors", + "model.layers.5.block_sparse_moe.gate.weight": "model-00006-of-00059.safetensors", + "model.layers.5.input_layernorm.weight": "model-00007-of-00059.safetensors", + "model.layers.5.post_attention_layernorm.weight": "model-00007-of-00059.safetensors", + "model.layers.5.self_attn.k_proj.weight": "model-00006-of-00059.safetensors", + "model.layers.5.self_attn.o_proj.weight": "model-00006-of-00059.safetensors", + "model.layers.5.self_attn.q_proj.weight": "model-00006-of-00059.safetensors", + "model.layers.5.self_attn.v_proj.weight": "model-00006-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.0.w1.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.0.w2.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.0.w3.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.1.w1.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.1.w2.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.1.w3.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.2.w1.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.2.w2.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.2.w3.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.3.w1.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.3.w2.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.3.w3.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.4.w1.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.4.w2.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.4.w3.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.5.w1.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.5.w2.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.5.w3.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.6.w1.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.6.w2.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.6.w3.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.7.w1.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.7.w2.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.experts.7.w3.weight": "model-00053-of-00059.safetensors", + "model.layers.50.block_sparse_moe.gate.weight": "model-00053-of-00059.safetensors", + "model.layers.50.input_layernorm.weight": "model-00053-of-00059.safetensors", + "model.layers.50.post_attention_layernorm.weight": "model-00053-of-00059.safetensors", + "model.layers.50.self_attn.k_proj.weight": "model-00052-of-00059.safetensors", + "model.layers.50.self_attn.o_proj.weight": "model-00053-of-00059.safetensors", + "model.layers.50.self_attn.q_proj.weight": "model-00052-of-00059.safetensors", + "model.layers.50.self_attn.v_proj.weight": "model-00052-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.0.w1.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.0.w2.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.0.w3.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.1.w1.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.1.w2.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.1.w3.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.2.w1.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.2.w2.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.2.w3.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.3.w1.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.3.w2.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.3.w3.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.4.w1.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.4.w2.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.4.w3.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.5.w1.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.5.w2.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.5.w3.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.6.w1.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.6.w2.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.6.w3.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.7.w1.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.7.w2.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.experts.7.w3.weight": "model-00054-of-00059.safetensors", + "model.layers.51.block_sparse_moe.gate.weight": "model-00054-of-00059.safetensors", + "model.layers.51.input_layernorm.weight": "model-00054-of-00059.safetensors", + "model.layers.51.post_attention_layernorm.weight": "model-00054-of-00059.safetensors", + "model.layers.51.self_attn.k_proj.weight": "model-00053-of-00059.safetensors", + "model.layers.51.self_attn.o_proj.weight": "model-00054-of-00059.safetensors", + "model.layers.51.self_attn.q_proj.weight": "model-00053-of-00059.safetensors", + "model.layers.51.self_attn.v_proj.weight": "model-00054-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.0.w1.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.0.w2.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.0.w3.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.1.w1.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.1.w2.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.1.w3.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.2.w1.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.2.w2.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.2.w3.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.3.w1.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.3.w2.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.3.w3.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.4.w1.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.4.w2.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.4.w3.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.5.w1.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.5.w2.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.5.w3.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.6.w1.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.6.w2.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.6.w3.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.7.w1.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.7.w2.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.experts.7.w3.weight": "model-00055-of-00059.safetensors", + "model.layers.52.block_sparse_moe.gate.weight": "model-00055-of-00059.safetensors", + "model.layers.52.input_layernorm.weight": "model-00055-of-00059.safetensors", + "model.layers.52.post_attention_layernorm.weight": "model-00055-of-00059.safetensors", + "model.layers.52.self_attn.k_proj.weight": "model-00055-of-00059.safetensors", + "model.layers.52.self_attn.o_proj.weight": "model-00055-of-00059.safetensors", + "model.layers.52.self_attn.q_proj.weight": "model-00054-of-00059.safetensors", + "model.layers.52.self_attn.v_proj.weight": "model-00055-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.0.w1.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.0.w2.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.0.w3.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.1.w1.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.1.w2.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.1.w3.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.2.w1.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.2.w2.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.2.w3.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.3.w1.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.3.w2.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.3.w3.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.4.w1.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.4.w2.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.4.w3.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.5.w1.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.5.w2.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.5.w3.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.6.w1.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.6.w2.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.6.w3.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.7.w1.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.7.w2.weight": "model-00056-of-00059.safetensors", + "model.layers.53.block_sparse_moe.experts.7.w3.weight": "model-00057-of-00059.safetensors", + "model.layers.53.block_sparse_moe.gate.weight": "model-00056-of-00059.safetensors", + "model.layers.53.input_layernorm.weight": "model-00057-of-00059.safetensors", + "model.layers.53.post_attention_layernorm.weight": "model-00057-of-00059.safetensors", + "model.layers.53.self_attn.k_proj.weight": "model-00056-of-00059.safetensors", + "model.layers.53.self_attn.o_proj.weight": "model-00056-of-00059.safetensors", + "model.layers.53.self_attn.q_proj.weight": "model-00056-of-00059.safetensors", + "model.layers.53.self_attn.v_proj.weight": "model-00056-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.0.w1.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.0.w2.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.0.w3.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.1.w1.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.1.w2.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.1.w3.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.2.w1.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.2.w2.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.2.w3.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.3.w1.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.3.w2.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.3.w3.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.4.w1.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.4.w2.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.4.w3.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.5.w1.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.5.w2.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.5.w3.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.6.w1.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.6.w2.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.6.w3.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.7.w1.weight": "model-00057-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.7.w2.weight": "model-00058-of-00059.safetensors", + "model.layers.54.block_sparse_moe.experts.7.w3.weight": "model-00058-of-00059.safetensors", + "model.layers.54.block_sparse_moe.gate.weight": "model-00057-of-00059.safetensors", + "model.layers.54.input_layernorm.weight": "model-00058-of-00059.safetensors", + "model.layers.54.post_attention_layernorm.weight": "model-00058-of-00059.safetensors", + "model.layers.54.self_attn.k_proj.weight": "model-00057-of-00059.safetensors", + "model.layers.54.self_attn.o_proj.weight": "model-00057-of-00059.safetensors", + "model.layers.54.self_attn.q_proj.weight": "model-00057-of-00059.safetensors", + "model.layers.54.self_attn.v_proj.weight": "model-00057-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.0.w1.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.0.w2.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.0.w3.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.1.w1.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.1.w2.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.1.w3.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.2.w1.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.2.w2.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.2.w3.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.3.w1.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.3.w2.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.3.w3.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.4.w1.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.4.w2.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.4.w3.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.5.w1.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.5.w2.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.5.w3.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.6.w1.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.6.w2.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.6.w3.weight": "model-00058-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.7.w1.weight": "model-00059-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.7.w2.weight": "model-00059-of-00059.safetensors", + "model.layers.55.block_sparse_moe.experts.7.w3.weight": "model-00059-of-00059.safetensors", + "model.layers.55.block_sparse_moe.gate.weight": "model-00058-of-00059.safetensors", + "model.layers.55.input_layernorm.weight": "model-00059-of-00059.safetensors", + "model.layers.55.post_attention_layernorm.weight": "model-00059-of-00059.safetensors", + "model.layers.55.self_attn.k_proj.weight": "model-00058-of-00059.safetensors", + "model.layers.55.self_attn.o_proj.weight": "model-00058-of-00059.safetensors", + "model.layers.55.self_attn.q_proj.weight": "model-00058-of-00059.safetensors", + "model.layers.55.self_attn.v_proj.weight": "model-00058-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.0.w1.weight": "model-00007-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.0.w2.weight": "model-00007-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.0.w3.weight": "model-00007-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.1.w1.weight": "model-00007-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.1.w2.weight": "model-00007-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.1.w3.weight": "model-00007-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.2.w1.weight": "model-00007-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.2.w2.weight": "model-00007-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.2.w3.weight": "model-00007-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.3.w1.weight": "model-00007-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.3.w2.weight": "model-00007-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.3.w3.weight": "model-00007-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.4.w1.weight": "model-00007-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.4.w2.weight": "model-00007-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.4.w3.weight": "model-00007-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.5.w1.weight": "model-00007-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.5.w2.weight": "model-00008-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.5.w3.weight": "model-00008-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.6.w1.weight": "model-00008-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.6.w2.weight": "model-00008-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.6.w3.weight": "model-00008-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.7.w1.weight": "model-00008-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.7.w2.weight": "model-00008-of-00059.safetensors", + "model.layers.6.block_sparse_moe.experts.7.w3.weight": "model-00008-of-00059.safetensors", + "model.layers.6.block_sparse_moe.gate.weight": "model-00007-of-00059.safetensors", + "model.layers.6.input_layernorm.weight": "model-00008-of-00059.safetensors", + "model.layers.6.post_attention_layernorm.weight": "model-00008-of-00059.safetensors", + "model.layers.6.self_attn.k_proj.weight": "model-00007-of-00059.safetensors", + "model.layers.6.self_attn.o_proj.weight": "model-00007-of-00059.safetensors", + "model.layers.6.self_attn.q_proj.weight": "model-00007-of-00059.safetensors", + "model.layers.6.self_attn.v_proj.weight": "model-00007-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.0.w1.weight": "model-00008-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.0.w2.weight": "model-00008-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.0.w3.weight": "model-00008-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.1.w1.weight": "model-00008-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.1.w2.weight": "model-00008-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.1.w3.weight": "model-00008-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.2.w1.weight": "model-00008-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.2.w2.weight": "model-00008-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.2.w3.weight": "model-00008-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.3.w1.weight": "model-00008-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.3.w2.weight": "model-00008-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.3.w3.weight": "model-00008-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.4.w1.weight": "model-00008-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.4.w2.weight": "model-00008-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.4.w3.weight": "model-00008-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.5.w1.weight": "model-00009-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.5.w2.weight": "model-00009-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.5.w3.weight": "model-00009-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.6.w1.weight": "model-00009-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.6.w2.weight": "model-00009-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.6.w3.weight": "model-00009-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.7.w1.weight": "model-00009-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.7.w2.weight": "model-00009-of-00059.safetensors", + "model.layers.7.block_sparse_moe.experts.7.w3.weight": "model-00009-of-00059.safetensors", + "model.layers.7.block_sparse_moe.gate.weight": "model-00008-of-00059.safetensors", + "model.layers.7.input_layernorm.weight": "model-00009-of-00059.safetensors", + "model.layers.7.post_attention_layernorm.weight": "model-00009-of-00059.safetensors", + "model.layers.7.self_attn.k_proj.weight": "model-00008-of-00059.safetensors", + "model.layers.7.self_attn.o_proj.weight": "model-00008-of-00059.safetensors", + "model.layers.7.self_attn.q_proj.weight": "model-00008-of-00059.safetensors", + "model.layers.7.self_attn.v_proj.weight": "model-00008-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.0.w1.weight": "model-00009-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.0.w2.weight": "model-00009-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.0.w3.weight": "model-00009-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.1.w1.weight": "model-00009-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.1.w2.weight": "model-00009-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.1.w3.weight": "model-00009-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.2.w1.weight": "model-00009-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.2.w2.weight": "model-00009-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.2.w3.weight": "model-00009-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.3.w1.weight": "model-00009-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.3.w2.weight": "model-00009-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.3.w3.weight": "model-00009-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.4.w1.weight": "model-00009-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.4.w2.weight": "model-00009-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.4.w3.weight": "model-00010-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.5.w1.weight": "model-00010-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.5.w2.weight": "model-00010-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.5.w3.weight": "model-00010-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.6.w1.weight": "model-00010-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.6.w2.weight": "model-00010-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.6.w3.weight": "model-00010-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.7.w1.weight": "model-00010-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.7.w2.weight": "model-00010-of-00059.safetensors", + "model.layers.8.block_sparse_moe.experts.7.w3.weight": "model-00010-of-00059.safetensors", + "model.layers.8.block_sparse_moe.gate.weight": "model-00009-of-00059.safetensors", + "model.layers.8.input_layernorm.weight": "model-00010-of-00059.safetensors", + "model.layers.8.post_attention_layernorm.weight": "model-00010-of-00059.safetensors", + "model.layers.8.self_attn.k_proj.weight": "model-00009-of-00059.safetensors", + "model.layers.8.self_attn.o_proj.weight": "model-00009-of-00059.safetensors", + "model.layers.8.self_attn.q_proj.weight": "model-00009-of-00059.safetensors", + "model.layers.8.self_attn.v_proj.weight": "model-00009-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.0.w1.weight": "model-00010-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.0.w2.weight": "model-00010-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.0.w3.weight": "model-00010-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.1.w1.weight": "model-00010-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.1.w2.weight": "model-00010-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.1.w3.weight": "model-00010-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.2.w1.weight": "model-00010-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.2.w2.weight": "model-00010-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.2.w3.weight": "model-00010-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.3.w1.weight": "model-00010-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.3.w2.weight": "model-00010-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.3.w3.weight": "model-00010-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.4.w1.weight": "model-00010-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.4.w2.weight": "model-00011-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.4.w3.weight": "model-00011-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.5.w1.weight": "model-00011-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.5.w2.weight": "model-00011-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.5.w3.weight": "model-00011-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.6.w1.weight": "model-00011-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.6.w2.weight": "model-00011-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.6.w3.weight": "model-00011-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.7.w1.weight": "model-00011-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.7.w2.weight": "model-00011-of-00059.safetensors", + "model.layers.9.block_sparse_moe.experts.7.w3.weight": "model-00011-of-00059.safetensors", + "model.layers.9.block_sparse_moe.gate.weight": "model-00010-of-00059.safetensors", + "model.layers.9.input_layernorm.weight": "model-00011-of-00059.safetensors", + "model.layers.9.post_attention_layernorm.weight": "model-00011-of-00059.safetensors", + "model.layers.9.self_attn.k_proj.weight": "model-00010-of-00059.safetensors", + "model.layers.9.self_attn.o_proj.weight": "model-00010-of-00059.safetensors", + "model.layers.9.self_attn.q_proj.weight": "model-00010-of-00059.safetensors", + "model.layers.9.self_attn.v_proj.weight": "model-00010-of-00059.safetensors", + "model.norm.weight": "model-00059-of-00059.safetensors" + } +} \ No newline at end of file