Upload folder using ModelScope SDK

This commit is contained in:
Cherrytest
2025-08-05 19:10:50 +00:00
parent 83440eb804
commit 0056bf5ec6
39 changed files with 1679 additions and 42 deletions

1
original/config.json Normal file
View File

@ -0,0 +1 @@
{"num_hidden_layers": 36, "num_experts": 128, "experts_per_token": 4, "vocab_size": 201088, "hidden_size": 2880, "intermediate_size": 2880, "swiglu_limit": 7.0, "head_dim": 64, "num_attention_heads": 64, "num_key_value_heads": 8, "sliding_window": 128, "initial_context_length": 4096, "rope_theta": 150000, "rope_scaling_factor": 32.0, "rope_ntk_alpha": 1, "rope_ntk_beta": 32}

1
original/dtypes.json Normal file

File diff suppressed because one or more lines are too long

View File

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:33e4fdf2f59f2a0d4846845a5140f675ad4599f959c2c496eed4fcac35a6b956
size 136

View File

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:8de8f2e9707625630e5285cda173412c7b631b5ddea9935d04f387dec277fa50
size 136

View File

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:8db53f7e257c08af4853d7cd1f4b77db68dd430b9a5aca61ac433b873ede7221
size 136

View File

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:f06b5622aca2074f1f8a80d92602f0d8b2bd4dd086ba3be5af72d7ba8db4a704
size 136

View File

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:85736f3aee44aea9a2208cb8e5b022121036d02355c63b97ce05e76098017f68
size 136

View File

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:c4211c260d9d588c70c1690a0e699435c86f90dab5cd77b46dfae3e110fcd9f0
size 136

View File

@ -0,0 +1,3 @@
version https://git-lfs.github.com/spec/v1
oid sha256:662393b6ddc88c257455b4dead5815c99f56589cdaa68576b396d5c92a5bb83c
size 135

View File

@ -0,0 +1,550 @@
{
"metadata": {
"total_size": 65248815744
},
"weight_map": {
"block.0.attn.norm.scale": "model--00001-of-00007.safetensors",
"block.0.attn.out.bias": "model--00001-of-00007.safetensors",
"block.0.attn.out.weight": "model--00001-of-00007.safetensors",
"block.0.attn.qkv.bias": "model--00001-of-00007.safetensors",
"block.0.attn.qkv.weight": "model--00001-of-00007.safetensors",
"block.0.attn.sinks": "model--00001-of-00007.safetensors",
"block.0.mlp.gate.bias": "model--00001-of-00007.safetensors",
"block.0.mlp.gate.weight": "model--00001-of-00007.safetensors",
"block.0.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
"block.0.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
"block.0.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
"block.0.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
"block.0.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
"block.0.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
"block.0.mlp.norm.scale": "model--00001-of-00007.safetensors",
"block.1.attn.norm.scale": "model--00001-of-00007.safetensors",
"block.1.attn.out.bias": "model--00001-of-00007.safetensors",
"block.1.attn.out.weight": "model--00001-of-00007.safetensors",
"block.1.attn.qkv.bias": "model--00001-of-00007.safetensors",
"block.1.attn.qkv.weight": "model--00001-of-00007.safetensors",
"block.1.attn.sinks": "model--00001-of-00007.safetensors",
"block.1.mlp.gate.bias": "model--00001-of-00007.safetensors",
"block.1.mlp.gate.weight": "model--00001-of-00007.safetensors",
"block.1.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
"block.1.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
"block.1.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
"block.1.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
"block.1.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
"block.1.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
"block.1.mlp.norm.scale": "model--00001-of-00007.safetensors",
"block.10.attn.norm.scale": "model--00001-of-00007.safetensors",
"block.10.attn.out.bias": "model--00001-of-00007.safetensors",
"block.10.attn.out.weight": "model--00001-of-00007.safetensors",
"block.10.attn.qkv.bias": "model--00001-of-00007.safetensors",
"block.10.attn.qkv.weight": "model--00001-of-00007.safetensors",
"block.10.attn.sinks": "model--00001-of-00007.safetensors",
"block.10.mlp.gate.bias": "model--00001-of-00007.safetensors",
"block.10.mlp.gate.weight": "model--00001-of-00007.safetensors",
"block.10.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
"block.10.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
"block.10.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
"block.10.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
"block.10.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
"block.10.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
"block.10.mlp.norm.scale": "model--00001-of-00007.safetensors",
"block.11.attn.norm.scale": "model--00001-of-00007.safetensors",
"block.11.attn.out.bias": "model--00001-of-00007.safetensors",
"block.11.attn.out.weight": "model--00001-of-00007.safetensors",
"block.11.attn.qkv.bias": "model--00001-of-00007.safetensors",
"block.11.attn.qkv.weight": "model--00001-of-00007.safetensors",
"block.11.attn.sinks": "model--00001-of-00007.safetensors",
"block.11.mlp.gate.bias": "model--00001-of-00007.safetensors",
"block.11.mlp.gate.weight": "model--00001-of-00007.safetensors",
"block.11.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
"block.11.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
"block.11.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
"block.11.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
"block.11.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
"block.11.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
"block.11.mlp.norm.scale": "model--00001-of-00007.safetensors",
"block.12.attn.norm.scale": "model--00001-of-00007.safetensors",
"block.12.attn.out.bias": "model--00001-of-00007.safetensors",
"block.12.attn.out.weight": "model--00001-of-00007.safetensors",
"block.12.attn.qkv.bias": "model--00001-of-00007.safetensors",
"block.12.attn.qkv.weight": "model--00001-of-00007.safetensors",
"block.12.attn.sinks": "model--00001-of-00007.safetensors",
"block.12.mlp.gate.bias": "model--00001-of-00007.safetensors",
"block.12.mlp.gate.weight": "model--00001-of-00007.safetensors",
"block.12.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
"block.12.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
"block.12.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
"block.12.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
"block.12.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
"block.12.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
"block.12.mlp.norm.scale": "model--00001-of-00007.safetensors",
"block.13.attn.norm.scale": "model--00001-of-00007.safetensors",
"block.13.attn.out.bias": "model--00001-of-00007.safetensors",
"block.13.attn.out.weight": "model--00001-of-00007.safetensors",
"block.13.attn.qkv.bias": "model--00001-of-00007.safetensors",
"block.13.attn.qkv.weight": "model--00001-of-00007.safetensors",
"block.13.attn.sinks": "model--00001-of-00007.safetensors",
"block.13.mlp.gate.bias": "model--00001-of-00007.safetensors",
"block.13.mlp.gate.weight": "model--00001-of-00007.safetensors",
"block.13.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
"block.13.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
"block.13.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
"block.13.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
"block.13.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
"block.13.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
"block.13.mlp.norm.scale": "model--00001-of-00007.safetensors",
"block.14.attn.norm.scale": "model--00001-of-00007.safetensors",
"block.14.attn.out.bias": "model--00001-of-00007.safetensors",
"block.14.attn.out.weight": "model--00001-of-00007.safetensors",
"block.14.attn.qkv.bias": "model--00001-of-00007.safetensors",
"block.14.attn.qkv.weight": "model--00001-of-00007.safetensors",
"block.14.attn.sinks": "model--00001-of-00007.safetensors",
"block.14.mlp.gate.bias": "model--00001-of-00007.safetensors",
"block.14.mlp.gate.weight": "model--00001-of-00007.safetensors",
"block.14.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
"block.14.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
"block.14.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
"block.14.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
"block.14.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
"block.14.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
"block.14.mlp.norm.scale": "model--00002-of-00007.safetensors",
"block.15.attn.norm.scale": "model--00002-of-00007.safetensors",
"block.15.attn.out.bias": "model--00002-of-00007.safetensors",
"block.15.attn.out.weight": "model--00002-of-00007.safetensors",
"block.15.attn.qkv.bias": "model--00002-of-00007.safetensors",
"block.15.attn.qkv.weight": "model--00002-of-00007.safetensors",
"block.15.attn.sinks": "model--00002-of-00007.safetensors",
"block.15.mlp.gate.bias": "model--00002-of-00007.safetensors",
"block.15.mlp.gate.weight": "model--00002-of-00007.safetensors",
"block.15.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
"block.15.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
"block.15.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
"block.15.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
"block.15.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
"block.15.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
"block.15.mlp.norm.scale": "model--00002-of-00007.safetensors",
"block.16.attn.norm.scale": "model--00002-of-00007.safetensors",
"block.16.attn.out.bias": "model--00002-of-00007.safetensors",
"block.16.attn.out.weight": "model--00002-of-00007.safetensors",
"block.16.attn.qkv.bias": "model--00002-of-00007.safetensors",
"block.16.attn.qkv.weight": "model--00002-of-00007.safetensors",
"block.16.attn.sinks": "model--00002-of-00007.safetensors",
"block.16.mlp.gate.bias": "model--00002-of-00007.safetensors",
"block.16.mlp.gate.weight": "model--00002-of-00007.safetensors",
"block.16.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
"block.16.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
"block.16.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
"block.16.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
"block.16.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
"block.16.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
"block.16.mlp.norm.scale": "model--00002-of-00007.safetensors",
"block.17.attn.norm.scale": "model--00002-of-00007.safetensors",
"block.17.attn.out.bias": "model--00002-of-00007.safetensors",
"block.17.attn.out.weight": "model--00002-of-00007.safetensors",
"block.17.attn.qkv.bias": "model--00002-of-00007.safetensors",
"block.17.attn.qkv.weight": "model--00002-of-00007.safetensors",
"block.17.attn.sinks": "model--00002-of-00007.safetensors",
"block.17.mlp.gate.bias": "model--00002-of-00007.safetensors",
"block.17.mlp.gate.weight": "model--00002-of-00007.safetensors",
"block.17.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
"block.17.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
"block.17.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
"block.17.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
"block.17.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
"block.17.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
"block.17.mlp.norm.scale": "model--00002-of-00007.safetensors",
"block.18.attn.norm.scale": "model--00002-of-00007.safetensors",
"block.18.attn.out.bias": "model--00002-of-00007.safetensors",
"block.18.attn.out.weight": "model--00002-of-00007.safetensors",
"block.18.attn.qkv.bias": "model--00002-of-00007.safetensors",
"block.18.attn.qkv.weight": "model--00002-of-00007.safetensors",
"block.18.attn.sinks": "model--00002-of-00007.safetensors",
"block.18.mlp.gate.bias": "model--00002-of-00007.safetensors",
"block.18.mlp.gate.weight": "model--00002-of-00007.safetensors",
"block.18.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
"block.18.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
"block.18.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
"block.18.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
"block.18.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
"block.18.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
"block.18.mlp.norm.scale": "model--00002-of-00007.safetensors",
"block.19.attn.norm.scale": "model--00002-of-00007.safetensors",
"block.19.attn.out.bias": "model--00002-of-00007.safetensors",
"block.19.attn.out.weight": "model--00002-of-00007.safetensors",
"block.19.attn.qkv.bias": "model--00002-of-00007.safetensors",
"block.19.attn.qkv.weight": "model--00002-of-00007.safetensors",
"block.19.attn.sinks": "model--00002-of-00007.safetensors",
"block.19.mlp.gate.bias": "model--00002-of-00007.safetensors",
"block.19.mlp.gate.weight": "model--00002-of-00007.safetensors",
"block.19.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
"block.19.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
"block.19.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
"block.19.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
"block.19.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
"block.19.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
"block.19.mlp.norm.scale": "model--00002-of-00007.safetensors",
"block.2.attn.norm.scale": "model--00002-of-00007.safetensors",
"block.2.attn.out.bias": "model--00002-of-00007.safetensors",
"block.2.attn.out.weight": "model--00002-of-00007.safetensors",
"block.2.attn.qkv.bias": "model--00002-of-00007.safetensors",
"block.2.attn.qkv.weight": "model--00002-of-00007.safetensors",
"block.2.attn.sinks": "model--00002-of-00007.safetensors",
"block.2.mlp.gate.bias": "model--00002-of-00007.safetensors",
"block.2.mlp.gate.weight": "model--00002-of-00007.safetensors",
"block.2.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
"block.2.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
"block.2.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
"block.2.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
"block.2.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
"block.2.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
"block.2.mlp.norm.scale": "model--00003-of-00007.safetensors",
"block.20.attn.norm.scale": "model--00003-of-00007.safetensors",
"block.20.attn.out.bias": "model--00003-of-00007.safetensors",
"block.20.attn.out.weight": "model--00003-of-00007.safetensors",
"block.20.attn.qkv.bias": "model--00003-of-00007.safetensors",
"block.20.attn.qkv.weight": "model--00003-of-00007.safetensors",
"block.20.attn.sinks": "model--00003-of-00007.safetensors",
"block.20.mlp.gate.bias": "model--00003-of-00007.safetensors",
"block.20.mlp.gate.weight": "model--00003-of-00007.safetensors",
"block.20.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
"block.20.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
"block.20.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
"block.20.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
"block.20.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
"block.20.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
"block.20.mlp.norm.scale": "model--00003-of-00007.safetensors",
"block.21.attn.norm.scale": "model--00003-of-00007.safetensors",
"block.21.attn.out.bias": "model--00003-of-00007.safetensors",
"block.21.attn.out.weight": "model--00003-of-00007.safetensors",
"block.21.attn.qkv.bias": "model--00003-of-00007.safetensors",
"block.21.attn.qkv.weight": "model--00003-of-00007.safetensors",
"block.21.attn.sinks": "model--00003-of-00007.safetensors",
"block.21.mlp.gate.bias": "model--00003-of-00007.safetensors",
"block.21.mlp.gate.weight": "model--00003-of-00007.safetensors",
"block.21.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
"block.21.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
"block.21.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
"block.21.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
"block.21.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
"block.21.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
"block.21.mlp.norm.scale": "model--00003-of-00007.safetensors",
"block.22.attn.norm.scale": "model--00003-of-00007.safetensors",
"block.22.attn.out.bias": "model--00003-of-00007.safetensors",
"block.22.attn.out.weight": "model--00003-of-00007.safetensors",
"block.22.attn.qkv.bias": "model--00003-of-00007.safetensors",
"block.22.attn.qkv.weight": "model--00003-of-00007.safetensors",
"block.22.attn.sinks": "model--00003-of-00007.safetensors",
"block.22.mlp.gate.bias": "model--00003-of-00007.safetensors",
"block.22.mlp.gate.weight": "model--00003-of-00007.safetensors",
"block.22.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
"block.22.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
"block.22.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
"block.22.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
"block.22.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
"block.22.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
"block.22.mlp.norm.scale": "model--00003-of-00007.safetensors",
"block.23.attn.norm.scale": "model--00003-of-00007.safetensors",
"block.23.attn.out.bias": "model--00003-of-00007.safetensors",
"block.23.attn.out.weight": "model--00003-of-00007.safetensors",
"block.23.attn.qkv.bias": "model--00003-of-00007.safetensors",
"block.23.attn.qkv.weight": "model--00003-of-00007.safetensors",
"block.23.attn.sinks": "model--00003-of-00007.safetensors",
"block.23.mlp.gate.bias": "model--00003-of-00007.safetensors",
"block.23.mlp.gate.weight": "model--00003-of-00007.safetensors",
"block.23.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
"block.23.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
"block.23.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
"block.23.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
"block.23.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
"block.23.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
"block.23.mlp.norm.scale": "model--00003-of-00007.safetensors",
"block.24.attn.norm.scale": "model--00003-of-00007.safetensors",
"block.24.attn.out.bias": "model--00003-of-00007.safetensors",
"block.24.attn.out.weight": "model--00003-of-00007.safetensors",
"block.24.attn.qkv.bias": "model--00003-of-00007.safetensors",
"block.24.attn.qkv.weight": "model--00003-of-00007.safetensors",
"block.24.attn.sinks": "model--00003-of-00007.safetensors",
"block.24.mlp.gate.bias": "model--00003-of-00007.safetensors",
"block.24.mlp.gate.weight": "model--00003-of-00007.safetensors",
"block.24.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
"block.24.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
"block.24.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
"block.24.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
"block.24.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
"block.24.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
"block.24.mlp.norm.scale": "model--00003-of-00007.safetensors",
"block.25.attn.norm.scale": "model--00003-of-00007.safetensors",
"block.25.attn.out.bias": "model--00003-of-00007.safetensors",
"block.25.attn.out.weight": "model--00003-of-00007.safetensors",
"block.25.attn.qkv.bias": "model--00003-of-00007.safetensors",
"block.25.attn.qkv.weight": "model--00003-of-00007.safetensors",
"block.25.attn.sinks": "model--00003-of-00007.safetensors",
"block.25.mlp.gate.bias": "model--00003-of-00007.safetensors",
"block.25.mlp.gate.weight": "model--00003-of-00007.safetensors",
"block.25.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
"block.25.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
"block.25.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
"block.25.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
"block.25.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
"block.25.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
"block.25.mlp.norm.scale": "model--00004-of-00007.safetensors",
"block.26.attn.norm.scale": "model--00004-of-00007.safetensors",
"block.26.attn.out.bias": "model--00004-of-00007.safetensors",
"block.26.attn.out.weight": "model--00004-of-00007.safetensors",
"block.26.attn.qkv.bias": "model--00004-of-00007.safetensors",
"block.26.attn.qkv.weight": "model--00004-of-00007.safetensors",
"block.26.attn.sinks": "model--00004-of-00007.safetensors",
"block.26.mlp.gate.bias": "model--00004-of-00007.safetensors",
"block.26.mlp.gate.weight": "model--00004-of-00007.safetensors",
"block.26.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
"block.26.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
"block.26.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
"block.26.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
"block.26.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
"block.26.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
"block.26.mlp.norm.scale": "model--00004-of-00007.safetensors",
"block.27.attn.norm.scale": "model--00004-of-00007.safetensors",
"block.27.attn.out.bias": "model--00004-of-00007.safetensors",
"block.27.attn.out.weight": "model--00004-of-00007.safetensors",
"block.27.attn.qkv.bias": "model--00004-of-00007.safetensors",
"block.27.attn.qkv.weight": "model--00004-of-00007.safetensors",
"block.27.attn.sinks": "model--00004-of-00007.safetensors",
"block.27.mlp.gate.bias": "model--00004-of-00007.safetensors",
"block.27.mlp.gate.weight": "model--00004-of-00007.safetensors",
"block.27.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
"block.27.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
"block.27.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
"block.27.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
"block.27.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
"block.27.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
"block.27.mlp.norm.scale": "model--00004-of-00007.safetensors",
"block.28.attn.norm.scale": "model--00004-of-00007.safetensors",
"block.28.attn.out.bias": "model--00004-of-00007.safetensors",
"block.28.attn.out.weight": "model--00004-of-00007.safetensors",
"block.28.attn.qkv.bias": "model--00004-of-00007.safetensors",
"block.28.attn.qkv.weight": "model--00004-of-00007.safetensors",
"block.28.attn.sinks": "model--00004-of-00007.safetensors",
"block.28.mlp.gate.bias": "model--00004-of-00007.safetensors",
"block.28.mlp.gate.weight": "model--00004-of-00007.safetensors",
"block.28.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
"block.28.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
"block.28.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
"block.28.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
"block.28.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
"block.28.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
"block.28.mlp.norm.scale": "model--00004-of-00007.safetensors",
"block.29.attn.norm.scale": "model--00004-of-00007.safetensors",
"block.29.attn.out.bias": "model--00004-of-00007.safetensors",
"block.29.attn.out.weight": "model--00004-of-00007.safetensors",
"block.29.attn.qkv.bias": "model--00004-of-00007.safetensors",
"block.29.attn.qkv.weight": "model--00004-of-00007.safetensors",
"block.29.attn.sinks": "model--00004-of-00007.safetensors",
"block.29.mlp.gate.bias": "model--00004-of-00007.safetensors",
"block.29.mlp.gate.weight": "model--00004-of-00007.safetensors",
"block.29.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
"block.29.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
"block.29.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
"block.29.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
"block.29.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
"block.29.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
"block.29.mlp.norm.scale": "model--00004-of-00007.safetensors",
"block.3.attn.norm.scale": "model--00004-of-00007.safetensors",
"block.3.attn.out.bias": "model--00004-of-00007.safetensors",
"block.3.attn.out.weight": "model--00004-of-00007.safetensors",
"block.3.attn.qkv.bias": "model--00004-of-00007.safetensors",
"block.3.attn.qkv.weight": "model--00004-of-00007.safetensors",
"block.3.attn.sinks": "model--00004-of-00007.safetensors",
"block.3.mlp.gate.bias": "model--00004-of-00007.safetensors",
"block.3.mlp.gate.weight": "model--00004-of-00007.safetensors",
"block.3.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
"block.3.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
"block.3.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
"block.3.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
"block.3.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
"block.3.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
"block.3.mlp.norm.scale": "model--00004-of-00007.safetensors",
"block.30.attn.norm.scale": "model--00004-of-00007.safetensors",
"block.30.attn.out.bias": "model--00004-of-00007.safetensors",
"block.30.attn.out.weight": "model--00004-of-00007.safetensors",
"block.30.attn.qkv.bias": "model--00004-of-00007.safetensors",
"block.30.attn.qkv.weight": "model--00004-of-00007.safetensors",
"block.30.attn.sinks": "model--00004-of-00007.safetensors",
"block.30.mlp.gate.bias": "model--00004-of-00007.safetensors",
"block.30.mlp.gate.weight": "model--00004-of-00007.safetensors",
"block.30.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
"block.30.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
"block.30.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
"block.30.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
"block.30.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
"block.30.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
"block.30.mlp.norm.scale": "model--00005-of-00007.safetensors",
"block.31.attn.norm.scale": "model--00005-of-00007.safetensors",
"block.31.attn.out.bias": "model--00005-of-00007.safetensors",
"block.31.attn.out.weight": "model--00005-of-00007.safetensors",
"block.31.attn.qkv.bias": "model--00005-of-00007.safetensors",
"block.31.attn.qkv.weight": "model--00005-of-00007.safetensors",
"block.31.attn.sinks": "model--00005-of-00007.safetensors",
"block.31.mlp.gate.bias": "model--00005-of-00007.safetensors",
"block.31.mlp.gate.weight": "model--00005-of-00007.safetensors",
"block.31.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
"block.31.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
"block.31.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
"block.31.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
"block.31.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
"block.31.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
"block.31.mlp.norm.scale": "model--00005-of-00007.safetensors",
"block.32.attn.norm.scale": "model--00005-of-00007.safetensors",
"block.32.attn.out.bias": "model--00005-of-00007.safetensors",
"block.32.attn.out.weight": "model--00005-of-00007.safetensors",
"block.32.attn.qkv.bias": "model--00005-of-00007.safetensors",
"block.32.attn.qkv.weight": "model--00005-of-00007.safetensors",
"block.32.attn.sinks": "model--00005-of-00007.safetensors",
"block.32.mlp.gate.bias": "model--00005-of-00007.safetensors",
"block.32.mlp.gate.weight": "model--00005-of-00007.safetensors",
"block.32.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
"block.32.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
"block.32.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
"block.32.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
"block.32.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
"block.32.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
"block.32.mlp.norm.scale": "model--00005-of-00007.safetensors",
"block.33.attn.norm.scale": "model--00005-of-00007.safetensors",
"block.33.attn.out.bias": "model--00005-of-00007.safetensors",
"block.33.attn.out.weight": "model--00005-of-00007.safetensors",
"block.33.attn.qkv.bias": "model--00005-of-00007.safetensors",
"block.33.attn.qkv.weight": "model--00005-of-00007.safetensors",
"block.33.attn.sinks": "model--00005-of-00007.safetensors",
"block.33.mlp.gate.bias": "model--00005-of-00007.safetensors",
"block.33.mlp.gate.weight": "model--00005-of-00007.safetensors",
"block.33.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
"block.33.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
"block.33.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
"block.33.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
"block.33.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
"block.33.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
"block.33.mlp.norm.scale": "model--00005-of-00007.safetensors",
"block.34.attn.norm.scale": "model--00005-of-00007.safetensors",
"block.34.attn.out.bias": "model--00005-of-00007.safetensors",
"block.34.attn.out.weight": "model--00005-of-00007.safetensors",
"block.34.attn.qkv.bias": "model--00005-of-00007.safetensors",
"block.34.attn.qkv.weight": "model--00005-of-00007.safetensors",
"block.34.attn.sinks": "model--00005-of-00007.safetensors",
"block.34.mlp.gate.bias": "model--00005-of-00007.safetensors",
"block.34.mlp.gate.weight": "model--00005-of-00007.safetensors",
"block.34.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
"block.34.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
"block.34.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
"block.34.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
"block.34.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
"block.34.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
"block.34.mlp.norm.scale": "model--00005-of-00007.safetensors",
"block.35.attn.norm.scale": "model--00005-of-00007.safetensors",
"block.35.attn.out.bias": "model--00005-of-00007.safetensors",
"block.35.attn.out.weight": "model--00005-of-00007.safetensors",
"block.35.attn.qkv.bias": "model--00005-of-00007.safetensors",
"block.35.attn.qkv.weight": "model--00005-of-00007.safetensors",
"block.35.attn.sinks": "model--00005-of-00007.safetensors",
"block.35.mlp.gate.bias": "model--00005-of-00007.safetensors",
"block.35.mlp.gate.weight": "model--00005-of-00007.safetensors",
"block.35.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
"block.35.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
"block.35.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
"block.35.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
"block.35.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
"block.35.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
"block.35.mlp.norm.scale": "model--00005-of-00007.safetensors",
"block.4.attn.norm.scale": "model--00005-of-00007.safetensors",
"block.4.attn.out.bias": "model--00005-of-00007.safetensors",
"block.4.attn.out.weight": "model--00005-of-00007.safetensors",
"block.4.attn.qkv.bias": "model--00005-of-00007.safetensors",
"block.4.attn.qkv.weight": "model--00005-of-00007.safetensors",
"block.4.attn.sinks": "model--00005-of-00007.safetensors",
"block.4.mlp.gate.bias": "model--00005-of-00007.safetensors",
"block.4.mlp.gate.weight": "model--00005-of-00007.safetensors",
"block.4.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
"block.4.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
"block.4.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
"block.4.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
"block.4.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
"block.4.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
"block.4.mlp.norm.scale": "model--00006-of-00007.safetensors",
"block.5.attn.norm.scale": "model--00006-of-00007.safetensors",
"block.5.attn.out.bias": "model--00006-of-00007.safetensors",
"block.5.attn.out.weight": "model--00006-of-00007.safetensors",
"block.5.attn.qkv.bias": "model--00006-of-00007.safetensors",
"block.5.attn.qkv.weight": "model--00006-of-00007.safetensors",
"block.5.attn.sinks": "model--00006-of-00007.safetensors",
"block.5.mlp.gate.bias": "model--00006-of-00007.safetensors",
"block.5.mlp.gate.weight": "model--00006-of-00007.safetensors",
"block.5.mlp.mlp1_bias": "model--00006-of-00007.safetensors",
"block.5.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
"block.5.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
"block.5.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
"block.5.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
"block.5.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
"block.5.mlp.norm.scale": "model--00006-of-00007.safetensors",
"block.6.attn.norm.scale": "model--00006-of-00007.safetensors",
"block.6.attn.out.bias": "model--00006-of-00007.safetensors",
"block.6.attn.out.weight": "model--00006-of-00007.safetensors",
"block.6.attn.qkv.bias": "model--00006-of-00007.safetensors",
"block.6.attn.qkv.weight": "model--00006-of-00007.safetensors",
"block.6.attn.sinks": "model--00006-of-00007.safetensors",
"block.6.mlp.gate.bias": "model--00006-of-00007.safetensors",
"block.6.mlp.gate.weight": "model--00006-of-00007.safetensors",
"block.6.mlp.mlp1_bias": "model--00006-of-00007.safetensors",
"block.6.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
"block.6.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
"block.6.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
"block.6.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
"block.6.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
"block.6.mlp.norm.scale": "model--00006-of-00007.safetensors",
"block.7.attn.norm.scale": "model--00006-of-00007.safetensors",
"block.7.attn.out.bias": "model--00006-of-00007.safetensors",
"block.7.attn.out.weight": "model--00006-of-00007.safetensors",
"block.7.attn.qkv.bias": "model--00006-of-00007.safetensors",
"block.7.attn.qkv.weight": "model--00006-of-00007.safetensors",
"block.7.attn.sinks": "model--00006-of-00007.safetensors",
"block.7.mlp.gate.bias": "model--00006-of-00007.safetensors",
"block.7.mlp.gate.weight": "model--00006-of-00007.safetensors",
"block.7.mlp.mlp1_bias": "model--00006-of-00007.safetensors",
"block.7.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
"block.7.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
"block.7.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
"block.7.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
"block.7.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
"block.7.mlp.norm.scale": "model--00006-of-00007.safetensors",
"block.8.attn.norm.scale": "model--00006-of-00007.safetensors",
"block.8.attn.out.bias": "model--00006-of-00007.safetensors",
"block.8.attn.out.weight": "model--00006-of-00007.safetensors",
"block.8.attn.qkv.bias": "model--00006-of-00007.safetensors",
"block.8.attn.qkv.weight": "model--00006-of-00007.safetensors",
"block.8.attn.sinks": "model--00006-of-00007.safetensors",
"block.8.mlp.gate.bias": "model--00006-of-00007.safetensors",
"block.8.mlp.gate.weight": "model--00006-of-00007.safetensors",
"block.8.mlp.mlp1_bias": "model--00006-of-00007.safetensors",
"block.8.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
"block.8.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
"block.8.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
"block.8.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
"block.8.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
"block.8.mlp.norm.scale": "model--00006-of-00007.safetensors",
"block.9.attn.norm.scale": "model--00006-of-00007.safetensors",
"block.9.attn.out.bias": "model--00006-of-00007.safetensors",
"block.9.attn.out.weight": "model--00006-of-00007.safetensors",
"block.9.attn.qkv.bias": "model--00006-of-00007.safetensors",
"block.9.attn.qkv.weight": "model--00006-of-00007.safetensors",
"block.9.attn.sinks": "model--00006-of-00007.safetensors",
"block.9.mlp.gate.bias": "model--00006-of-00007.safetensors",
"block.9.mlp.gate.weight": "model--00006-of-00007.safetensors",
"block.9.mlp.mlp1_bias": "model--00006-of-00007.safetensors",
"block.9.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
"block.9.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
"block.9.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
"block.9.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
"block.9.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
"block.9.mlp.norm.scale": "model--00006-of-00007.safetensors",
"embedding.weight": "model--00007-of-00007.safetensors",
"norm.scale": "model--00007-of-00007.safetensors",
"unembedding.weight": "model--00007-of-00007.safetensors"
}
}