mirror of
https://www.modelscope.cn/openai-mirror/gpt-oss-120b.git
synced 2026-04-02 18:12:56 +08:00
Upload folder using ModelScope SDK
This commit is contained in:
1
original/config.json
Normal file
1
original/config.json
Normal file
@ -0,0 +1 @@
|
||||
{"num_hidden_layers": 36, "num_experts": 128, "experts_per_token": 4, "vocab_size": 201088, "hidden_size": 2880, "intermediate_size": 2880, "swiglu_limit": 7.0, "head_dim": 64, "num_attention_heads": 64, "num_key_value_heads": 8, "sliding_window": 128, "initial_context_length": 4096, "rope_theta": 150000, "rope_scaling_factor": 32.0, "rope_ntk_alpha": 1, "rope_ntk_beta": 32}
|
||||
1
original/dtypes.json
Normal file
1
original/dtypes.json
Normal file
File diff suppressed because one or more lines are too long
3
original/model--00001-of-00007.safetensors
Normal file
3
original/model--00001-of-00007.safetensors
Normal file
@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:33e4fdf2f59f2a0d4846845a5140f675ad4599f959c2c496eed4fcac35a6b956
|
||||
size 136
|
||||
3
original/model--00002-of-00007.safetensors
Normal file
3
original/model--00002-of-00007.safetensors
Normal file
@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:8de8f2e9707625630e5285cda173412c7b631b5ddea9935d04f387dec277fa50
|
||||
size 136
|
||||
3
original/model--00003-of-00007.safetensors
Normal file
3
original/model--00003-of-00007.safetensors
Normal file
@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:8db53f7e257c08af4853d7cd1f4b77db68dd430b9a5aca61ac433b873ede7221
|
||||
size 136
|
||||
3
original/model--00004-of-00007.safetensors
Normal file
3
original/model--00004-of-00007.safetensors
Normal file
@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:f06b5622aca2074f1f8a80d92602f0d8b2bd4dd086ba3be5af72d7ba8db4a704
|
||||
size 136
|
||||
3
original/model--00005-of-00007.safetensors
Normal file
3
original/model--00005-of-00007.safetensors
Normal file
@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:85736f3aee44aea9a2208cb8e5b022121036d02355c63b97ce05e76098017f68
|
||||
size 136
|
||||
3
original/model--00006-of-00007.safetensors
Normal file
3
original/model--00006-of-00007.safetensors
Normal file
@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:c4211c260d9d588c70c1690a0e699435c86f90dab5cd77b46dfae3e110fcd9f0
|
||||
size 136
|
||||
3
original/model--00007-of-00007.safetensors
Normal file
3
original/model--00007-of-00007.safetensors
Normal file
@ -0,0 +1,3 @@
|
||||
version https://git-lfs.github.com/spec/v1
|
||||
oid sha256:662393b6ddc88c257455b4dead5815c99f56589cdaa68576b396d5c92a5bb83c
|
||||
size 135
|
||||
550
original/model.safetensors.index.json
Normal file
550
original/model.safetensors.index.json
Normal file
@ -0,0 +1,550 @@
|
||||
{
|
||||
"metadata": {
|
||||
"total_size": 65248815744
|
||||
},
|
||||
"weight_map": {
|
||||
"block.0.attn.norm.scale": "model--00001-of-00007.safetensors",
|
||||
"block.0.attn.out.bias": "model--00001-of-00007.safetensors",
|
||||
"block.0.attn.out.weight": "model--00001-of-00007.safetensors",
|
||||
"block.0.attn.qkv.bias": "model--00001-of-00007.safetensors",
|
||||
"block.0.attn.qkv.weight": "model--00001-of-00007.safetensors",
|
||||
"block.0.attn.sinks": "model--00001-of-00007.safetensors",
|
||||
"block.0.mlp.gate.bias": "model--00001-of-00007.safetensors",
|
||||
"block.0.mlp.gate.weight": "model--00001-of-00007.safetensors",
|
||||
"block.0.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
|
||||
"block.0.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
|
||||
"block.0.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
|
||||
"block.0.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
|
||||
"block.0.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
|
||||
"block.0.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
|
||||
"block.0.mlp.norm.scale": "model--00001-of-00007.safetensors",
|
||||
"block.1.attn.norm.scale": "model--00001-of-00007.safetensors",
|
||||
"block.1.attn.out.bias": "model--00001-of-00007.safetensors",
|
||||
"block.1.attn.out.weight": "model--00001-of-00007.safetensors",
|
||||
"block.1.attn.qkv.bias": "model--00001-of-00007.safetensors",
|
||||
"block.1.attn.qkv.weight": "model--00001-of-00007.safetensors",
|
||||
"block.1.attn.sinks": "model--00001-of-00007.safetensors",
|
||||
"block.1.mlp.gate.bias": "model--00001-of-00007.safetensors",
|
||||
"block.1.mlp.gate.weight": "model--00001-of-00007.safetensors",
|
||||
"block.1.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
|
||||
"block.1.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
|
||||
"block.1.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
|
||||
"block.1.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
|
||||
"block.1.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
|
||||
"block.1.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
|
||||
"block.1.mlp.norm.scale": "model--00001-of-00007.safetensors",
|
||||
"block.10.attn.norm.scale": "model--00001-of-00007.safetensors",
|
||||
"block.10.attn.out.bias": "model--00001-of-00007.safetensors",
|
||||
"block.10.attn.out.weight": "model--00001-of-00007.safetensors",
|
||||
"block.10.attn.qkv.bias": "model--00001-of-00007.safetensors",
|
||||
"block.10.attn.qkv.weight": "model--00001-of-00007.safetensors",
|
||||
"block.10.attn.sinks": "model--00001-of-00007.safetensors",
|
||||
"block.10.mlp.gate.bias": "model--00001-of-00007.safetensors",
|
||||
"block.10.mlp.gate.weight": "model--00001-of-00007.safetensors",
|
||||
"block.10.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
|
||||
"block.10.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
|
||||
"block.10.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
|
||||
"block.10.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
|
||||
"block.10.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
|
||||
"block.10.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
|
||||
"block.10.mlp.norm.scale": "model--00001-of-00007.safetensors",
|
||||
"block.11.attn.norm.scale": "model--00001-of-00007.safetensors",
|
||||
"block.11.attn.out.bias": "model--00001-of-00007.safetensors",
|
||||
"block.11.attn.out.weight": "model--00001-of-00007.safetensors",
|
||||
"block.11.attn.qkv.bias": "model--00001-of-00007.safetensors",
|
||||
"block.11.attn.qkv.weight": "model--00001-of-00007.safetensors",
|
||||
"block.11.attn.sinks": "model--00001-of-00007.safetensors",
|
||||
"block.11.mlp.gate.bias": "model--00001-of-00007.safetensors",
|
||||
"block.11.mlp.gate.weight": "model--00001-of-00007.safetensors",
|
||||
"block.11.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
|
||||
"block.11.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
|
||||
"block.11.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
|
||||
"block.11.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
|
||||
"block.11.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
|
||||
"block.11.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
|
||||
"block.11.mlp.norm.scale": "model--00001-of-00007.safetensors",
|
||||
"block.12.attn.norm.scale": "model--00001-of-00007.safetensors",
|
||||
"block.12.attn.out.bias": "model--00001-of-00007.safetensors",
|
||||
"block.12.attn.out.weight": "model--00001-of-00007.safetensors",
|
||||
"block.12.attn.qkv.bias": "model--00001-of-00007.safetensors",
|
||||
"block.12.attn.qkv.weight": "model--00001-of-00007.safetensors",
|
||||
"block.12.attn.sinks": "model--00001-of-00007.safetensors",
|
||||
"block.12.mlp.gate.bias": "model--00001-of-00007.safetensors",
|
||||
"block.12.mlp.gate.weight": "model--00001-of-00007.safetensors",
|
||||
"block.12.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
|
||||
"block.12.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
|
||||
"block.12.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
|
||||
"block.12.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
|
||||
"block.12.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
|
||||
"block.12.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
|
||||
"block.12.mlp.norm.scale": "model--00001-of-00007.safetensors",
|
||||
"block.13.attn.norm.scale": "model--00001-of-00007.safetensors",
|
||||
"block.13.attn.out.bias": "model--00001-of-00007.safetensors",
|
||||
"block.13.attn.out.weight": "model--00001-of-00007.safetensors",
|
||||
"block.13.attn.qkv.bias": "model--00001-of-00007.safetensors",
|
||||
"block.13.attn.qkv.weight": "model--00001-of-00007.safetensors",
|
||||
"block.13.attn.sinks": "model--00001-of-00007.safetensors",
|
||||
"block.13.mlp.gate.bias": "model--00001-of-00007.safetensors",
|
||||
"block.13.mlp.gate.weight": "model--00001-of-00007.safetensors",
|
||||
"block.13.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
|
||||
"block.13.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
|
||||
"block.13.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
|
||||
"block.13.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
|
||||
"block.13.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
|
||||
"block.13.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
|
||||
"block.13.mlp.norm.scale": "model--00001-of-00007.safetensors",
|
||||
"block.14.attn.norm.scale": "model--00001-of-00007.safetensors",
|
||||
"block.14.attn.out.bias": "model--00001-of-00007.safetensors",
|
||||
"block.14.attn.out.weight": "model--00001-of-00007.safetensors",
|
||||
"block.14.attn.qkv.bias": "model--00001-of-00007.safetensors",
|
||||
"block.14.attn.qkv.weight": "model--00001-of-00007.safetensors",
|
||||
"block.14.attn.sinks": "model--00001-of-00007.safetensors",
|
||||
"block.14.mlp.gate.bias": "model--00001-of-00007.safetensors",
|
||||
"block.14.mlp.gate.weight": "model--00001-of-00007.safetensors",
|
||||
"block.14.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
|
||||
"block.14.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
|
||||
"block.14.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
|
||||
"block.14.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
|
||||
"block.14.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
|
||||
"block.14.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
|
||||
"block.14.mlp.norm.scale": "model--00002-of-00007.safetensors",
|
||||
"block.15.attn.norm.scale": "model--00002-of-00007.safetensors",
|
||||
"block.15.attn.out.bias": "model--00002-of-00007.safetensors",
|
||||
"block.15.attn.out.weight": "model--00002-of-00007.safetensors",
|
||||
"block.15.attn.qkv.bias": "model--00002-of-00007.safetensors",
|
||||
"block.15.attn.qkv.weight": "model--00002-of-00007.safetensors",
|
||||
"block.15.attn.sinks": "model--00002-of-00007.safetensors",
|
||||
"block.15.mlp.gate.bias": "model--00002-of-00007.safetensors",
|
||||
"block.15.mlp.gate.weight": "model--00002-of-00007.safetensors",
|
||||
"block.15.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
|
||||
"block.15.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
|
||||
"block.15.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
|
||||
"block.15.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
|
||||
"block.15.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
|
||||
"block.15.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
|
||||
"block.15.mlp.norm.scale": "model--00002-of-00007.safetensors",
|
||||
"block.16.attn.norm.scale": "model--00002-of-00007.safetensors",
|
||||
"block.16.attn.out.bias": "model--00002-of-00007.safetensors",
|
||||
"block.16.attn.out.weight": "model--00002-of-00007.safetensors",
|
||||
"block.16.attn.qkv.bias": "model--00002-of-00007.safetensors",
|
||||
"block.16.attn.qkv.weight": "model--00002-of-00007.safetensors",
|
||||
"block.16.attn.sinks": "model--00002-of-00007.safetensors",
|
||||
"block.16.mlp.gate.bias": "model--00002-of-00007.safetensors",
|
||||
"block.16.mlp.gate.weight": "model--00002-of-00007.safetensors",
|
||||
"block.16.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
|
||||
"block.16.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
|
||||
"block.16.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
|
||||
"block.16.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
|
||||
"block.16.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
|
||||
"block.16.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
|
||||
"block.16.mlp.norm.scale": "model--00002-of-00007.safetensors",
|
||||
"block.17.attn.norm.scale": "model--00002-of-00007.safetensors",
|
||||
"block.17.attn.out.bias": "model--00002-of-00007.safetensors",
|
||||
"block.17.attn.out.weight": "model--00002-of-00007.safetensors",
|
||||
"block.17.attn.qkv.bias": "model--00002-of-00007.safetensors",
|
||||
"block.17.attn.qkv.weight": "model--00002-of-00007.safetensors",
|
||||
"block.17.attn.sinks": "model--00002-of-00007.safetensors",
|
||||
"block.17.mlp.gate.bias": "model--00002-of-00007.safetensors",
|
||||
"block.17.mlp.gate.weight": "model--00002-of-00007.safetensors",
|
||||
"block.17.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
|
||||
"block.17.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
|
||||
"block.17.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
|
||||
"block.17.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
|
||||
"block.17.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
|
||||
"block.17.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
|
||||
"block.17.mlp.norm.scale": "model--00002-of-00007.safetensors",
|
||||
"block.18.attn.norm.scale": "model--00002-of-00007.safetensors",
|
||||
"block.18.attn.out.bias": "model--00002-of-00007.safetensors",
|
||||
"block.18.attn.out.weight": "model--00002-of-00007.safetensors",
|
||||
"block.18.attn.qkv.bias": "model--00002-of-00007.safetensors",
|
||||
"block.18.attn.qkv.weight": "model--00002-of-00007.safetensors",
|
||||
"block.18.attn.sinks": "model--00002-of-00007.safetensors",
|
||||
"block.18.mlp.gate.bias": "model--00002-of-00007.safetensors",
|
||||
"block.18.mlp.gate.weight": "model--00002-of-00007.safetensors",
|
||||
"block.18.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
|
||||
"block.18.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
|
||||
"block.18.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
|
||||
"block.18.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
|
||||
"block.18.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
|
||||
"block.18.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
|
||||
"block.18.mlp.norm.scale": "model--00002-of-00007.safetensors",
|
||||
"block.19.attn.norm.scale": "model--00002-of-00007.safetensors",
|
||||
"block.19.attn.out.bias": "model--00002-of-00007.safetensors",
|
||||
"block.19.attn.out.weight": "model--00002-of-00007.safetensors",
|
||||
"block.19.attn.qkv.bias": "model--00002-of-00007.safetensors",
|
||||
"block.19.attn.qkv.weight": "model--00002-of-00007.safetensors",
|
||||
"block.19.attn.sinks": "model--00002-of-00007.safetensors",
|
||||
"block.19.mlp.gate.bias": "model--00002-of-00007.safetensors",
|
||||
"block.19.mlp.gate.weight": "model--00002-of-00007.safetensors",
|
||||
"block.19.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
|
||||
"block.19.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
|
||||
"block.19.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
|
||||
"block.19.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
|
||||
"block.19.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
|
||||
"block.19.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
|
||||
"block.19.mlp.norm.scale": "model--00002-of-00007.safetensors",
|
||||
"block.2.attn.norm.scale": "model--00002-of-00007.safetensors",
|
||||
"block.2.attn.out.bias": "model--00002-of-00007.safetensors",
|
||||
"block.2.attn.out.weight": "model--00002-of-00007.safetensors",
|
||||
"block.2.attn.qkv.bias": "model--00002-of-00007.safetensors",
|
||||
"block.2.attn.qkv.weight": "model--00002-of-00007.safetensors",
|
||||
"block.2.attn.sinks": "model--00002-of-00007.safetensors",
|
||||
"block.2.mlp.gate.bias": "model--00002-of-00007.safetensors",
|
||||
"block.2.mlp.gate.weight": "model--00002-of-00007.safetensors",
|
||||
"block.2.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
|
||||
"block.2.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
|
||||
"block.2.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
|
||||
"block.2.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
|
||||
"block.2.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
|
||||
"block.2.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
|
||||
"block.2.mlp.norm.scale": "model--00003-of-00007.safetensors",
|
||||
"block.20.attn.norm.scale": "model--00003-of-00007.safetensors",
|
||||
"block.20.attn.out.bias": "model--00003-of-00007.safetensors",
|
||||
"block.20.attn.out.weight": "model--00003-of-00007.safetensors",
|
||||
"block.20.attn.qkv.bias": "model--00003-of-00007.safetensors",
|
||||
"block.20.attn.qkv.weight": "model--00003-of-00007.safetensors",
|
||||
"block.20.attn.sinks": "model--00003-of-00007.safetensors",
|
||||
"block.20.mlp.gate.bias": "model--00003-of-00007.safetensors",
|
||||
"block.20.mlp.gate.weight": "model--00003-of-00007.safetensors",
|
||||
"block.20.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
|
||||
"block.20.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
|
||||
"block.20.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
|
||||
"block.20.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
|
||||
"block.20.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
|
||||
"block.20.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
|
||||
"block.20.mlp.norm.scale": "model--00003-of-00007.safetensors",
|
||||
"block.21.attn.norm.scale": "model--00003-of-00007.safetensors",
|
||||
"block.21.attn.out.bias": "model--00003-of-00007.safetensors",
|
||||
"block.21.attn.out.weight": "model--00003-of-00007.safetensors",
|
||||
"block.21.attn.qkv.bias": "model--00003-of-00007.safetensors",
|
||||
"block.21.attn.qkv.weight": "model--00003-of-00007.safetensors",
|
||||
"block.21.attn.sinks": "model--00003-of-00007.safetensors",
|
||||
"block.21.mlp.gate.bias": "model--00003-of-00007.safetensors",
|
||||
"block.21.mlp.gate.weight": "model--00003-of-00007.safetensors",
|
||||
"block.21.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
|
||||
"block.21.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
|
||||
"block.21.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
|
||||
"block.21.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
|
||||
"block.21.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
|
||||
"block.21.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
|
||||
"block.21.mlp.norm.scale": "model--00003-of-00007.safetensors",
|
||||
"block.22.attn.norm.scale": "model--00003-of-00007.safetensors",
|
||||
"block.22.attn.out.bias": "model--00003-of-00007.safetensors",
|
||||
"block.22.attn.out.weight": "model--00003-of-00007.safetensors",
|
||||
"block.22.attn.qkv.bias": "model--00003-of-00007.safetensors",
|
||||
"block.22.attn.qkv.weight": "model--00003-of-00007.safetensors",
|
||||
"block.22.attn.sinks": "model--00003-of-00007.safetensors",
|
||||
"block.22.mlp.gate.bias": "model--00003-of-00007.safetensors",
|
||||
"block.22.mlp.gate.weight": "model--00003-of-00007.safetensors",
|
||||
"block.22.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
|
||||
"block.22.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
|
||||
"block.22.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
|
||||
"block.22.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
|
||||
"block.22.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
|
||||
"block.22.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
|
||||
"block.22.mlp.norm.scale": "model--00003-of-00007.safetensors",
|
||||
"block.23.attn.norm.scale": "model--00003-of-00007.safetensors",
|
||||
"block.23.attn.out.bias": "model--00003-of-00007.safetensors",
|
||||
"block.23.attn.out.weight": "model--00003-of-00007.safetensors",
|
||||
"block.23.attn.qkv.bias": "model--00003-of-00007.safetensors",
|
||||
"block.23.attn.qkv.weight": "model--00003-of-00007.safetensors",
|
||||
"block.23.attn.sinks": "model--00003-of-00007.safetensors",
|
||||
"block.23.mlp.gate.bias": "model--00003-of-00007.safetensors",
|
||||
"block.23.mlp.gate.weight": "model--00003-of-00007.safetensors",
|
||||
"block.23.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
|
||||
"block.23.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
|
||||
"block.23.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
|
||||
"block.23.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
|
||||
"block.23.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
|
||||
"block.23.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
|
||||
"block.23.mlp.norm.scale": "model--00003-of-00007.safetensors",
|
||||
"block.24.attn.norm.scale": "model--00003-of-00007.safetensors",
|
||||
"block.24.attn.out.bias": "model--00003-of-00007.safetensors",
|
||||
"block.24.attn.out.weight": "model--00003-of-00007.safetensors",
|
||||
"block.24.attn.qkv.bias": "model--00003-of-00007.safetensors",
|
||||
"block.24.attn.qkv.weight": "model--00003-of-00007.safetensors",
|
||||
"block.24.attn.sinks": "model--00003-of-00007.safetensors",
|
||||
"block.24.mlp.gate.bias": "model--00003-of-00007.safetensors",
|
||||
"block.24.mlp.gate.weight": "model--00003-of-00007.safetensors",
|
||||
"block.24.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
|
||||
"block.24.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
|
||||
"block.24.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
|
||||
"block.24.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
|
||||
"block.24.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
|
||||
"block.24.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
|
||||
"block.24.mlp.norm.scale": "model--00003-of-00007.safetensors",
|
||||
"block.25.attn.norm.scale": "model--00003-of-00007.safetensors",
|
||||
"block.25.attn.out.bias": "model--00003-of-00007.safetensors",
|
||||
"block.25.attn.out.weight": "model--00003-of-00007.safetensors",
|
||||
"block.25.attn.qkv.bias": "model--00003-of-00007.safetensors",
|
||||
"block.25.attn.qkv.weight": "model--00003-of-00007.safetensors",
|
||||
"block.25.attn.sinks": "model--00003-of-00007.safetensors",
|
||||
"block.25.mlp.gate.bias": "model--00003-of-00007.safetensors",
|
||||
"block.25.mlp.gate.weight": "model--00003-of-00007.safetensors",
|
||||
"block.25.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
|
||||
"block.25.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
|
||||
"block.25.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
|
||||
"block.25.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
|
||||
"block.25.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
|
||||
"block.25.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
|
||||
"block.25.mlp.norm.scale": "model--00004-of-00007.safetensors",
|
||||
"block.26.attn.norm.scale": "model--00004-of-00007.safetensors",
|
||||
"block.26.attn.out.bias": "model--00004-of-00007.safetensors",
|
||||
"block.26.attn.out.weight": "model--00004-of-00007.safetensors",
|
||||
"block.26.attn.qkv.bias": "model--00004-of-00007.safetensors",
|
||||
"block.26.attn.qkv.weight": "model--00004-of-00007.safetensors",
|
||||
"block.26.attn.sinks": "model--00004-of-00007.safetensors",
|
||||
"block.26.mlp.gate.bias": "model--00004-of-00007.safetensors",
|
||||
"block.26.mlp.gate.weight": "model--00004-of-00007.safetensors",
|
||||
"block.26.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
|
||||
"block.26.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
|
||||
"block.26.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
|
||||
"block.26.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
|
||||
"block.26.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
|
||||
"block.26.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
|
||||
"block.26.mlp.norm.scale": "model--00004-of-00007.safetensors",
|
||||
"block.27.attn.norm.scale": "model--00004-of-00007.safetensors",
|
||||
"block.27.attn.out.bias": "model--00004-of-00007.safetensors",
|
||||
"block.27.attn.out.weight": "model--00004-of-00007.safetensors",
|
||||
"block.27.attn.qkv.bias": "model--00004-of-00007.safetensors",
|
||||
"block.27.attn.qkv.weight": "model--00004-of-00007.safetensors",
|
||||
"block.27.attn.sinks": "model--00004-of-00007.safetensors",
|
||||
"block.27.mlp.gate.bias": "model--00004-of-00007.safetensors",
|
||||
"block.27.mlp.gate.weight": "model--00004-of-00007.safetensors",
|
||||
"block.27.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
|
||||
"block.27.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
|
||||
"block.27.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
|
||||
"block.27.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
|
||||
"block.27.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
|
||||
"block.27.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
|
||||
"block.27.mlp.norm.scale": "model--00004-of-00007.safetensors",
|
||||
"block.28.attn.norm.scale": "model--00004-of-00007.safetensors",
|
||||
"block.28.attn.out.bias": "model--00004-of-00007.safetensors",
|
||||
"block.28.attn.out.weight": "model--00004-of-00007.safetensors",
|
||||
"block.28.attn.qkv.bias": "model--00004-of-00007.safetensors",
|
||||
"block.28.attn.qkv.weight": "model--00004-of-00007.safetensors",
|
||||
"block.28.attn.sinks": "model--00004-of-00007.safetensors",
|
||||
"block.28.mlp.gate.bias": "model--00004-of-00007.safetensors",
|
||||
"block.28.mlp.gate.weight": "model--00004-of-00007.safetensors",
|
||||
"block.28.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
|
||||
"block.28.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
|
||||
"block.28.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
|
||||
"block.28.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
|
||||
"block.28.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
|
||||
"block.28.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
|
||||
"block.28.mlp.norm.scale": "model--00004-of-00007.safetensors",
|
||||
"block.29.attn.norm.scale": "model--00004-of-00007.safetensors",
|
||||
"block.29.attn.out.bias": "model--00004-of-00007.safetensors",
|
||||
"block.29.attn.out.weight": "model--00004-of-00007.safetensors",
|
||||
"block.29.attn.qkv.bias": "model--00004-of-00007.safetensors",
|
||||
"block.29.attn.qkv.weight": "model--00004-of-00007.safetensors",
|
||||
"block.29.attn.sinks": "model--00004-of-00007.safetensors",
|
||||
"block.29.mlp.gate.bias": "model--00004-of-00007.safetensors",
|
||||
"block.29.mlp.gate.weight": "model--00004-of-00007.safetensors",
|
||||
"block.29.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
|
||||
"block.29.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
|
||||
"block.29.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
|
||||
"block.29.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
|
||||
"block.29.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
|
||||
"block.29.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
|
||||
"block.29.mlp.norm.scale": "model--00004-of-00007.safetensors",
|
||||
"block.3.attn.norm.scale": "model--00004-of-00007.safetensors",
|
||||
"block.3.attn.out.bias": "model--00004-of-00007.safetensors",
|
||||
"block.3.attn.out.weight": "model--00004-of-00007.safetensors",
|
||||
"block.3.attn.qkv.bias": "model--00004-of-00007.safetensors",
|
||||
"block.3.attn.qkv.weight": "model--00004-of-00007.safetensors",
|
||||
"block.3.attn.sinks": "model--00004-of-00007.safetensors",
|
||||
"block.3.mlp.gate.bias": "model--00004-of-00007.safetensors",
|
||||
"block.3.mlp.gate.weight": "model--00004-of-00007.safetensors",
|
||||
"block.3.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
|
||||
"block.3.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
|
||||
"block.3.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
|
||||
"block.3.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
|
||||
"block.3.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
|
||||
"block.3.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
|
||||
"block.3.mlp.norm.scale": "model--00004-of-00007.safetensors",
|
||||
"block.30.attn.norm.scale": "model--00004-of-00007.safetensors",
|
||||
"block.30.attn.out.bias": "model--00004-of-00007.safetensors",
|
||||
"block.30.attn.out.weight": "model--00004-of-00007.safetensors",
|
||||
"block.30.attn.qkv.bias": "model--00004-of-00007.safetensors",
|
||||
"block.30.attn.qkv.weight": "model--00004-of-00007.safetensors",
|
||||
"block.30.attn.sinks": "model--00004-of-00007.safetensors",
|
||||
"block.30.mlp.gate.bias": "model--00004-of-00007.safetensors",
|
||||
"block.30.mlp.gate.weight": "model--00004-of-00007.safetensors",
|
||||
"block.30.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
|
||||
"block.30.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
|
||||
"block.30.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
|
||||
"block.30.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
|
||||
"block.30.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
|
||||
"block.30.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
|
||||
"block.30.mlp.norm.scale": "model--00005-of-00007.safetensors",
|
||||
"block.31.attn.norm.scale": "model--00005-of-00007.safetensors",
|
||||
"block.31.attn.out.bias": "model--00005-of-00007.safetensors",
|
||||
"block.31.attn.out.weight": "model--00005-of-00007.safetensors",
|
||||
"block.31.attn.qkv.bias": "model--00005-of-00007.safetensors",
|
||||
"block.31.attn.qkv.weight": "model--00005-of-00007.safetensors",
|
||||
"block.31.attn.sinks": "model--00005-of-00007.safetensors",
|
||||
"block.31.mlp.gate.bias": "model--00005-of-00007.safetensors",
|
||||
"block.31.mlp.gate.weight": "model--00005-of-00007.safetensors",
|
||||
"block.31.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
|
||||
"block.31.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
|
||||
"block.31.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
|
||||
"block.31.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
|
||||
"block.31.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
|
||||
"block.31.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
|
||||
"block.31.mlp.norm.scale": "model--00005-of-00007.safetensors",
|
||||
"block.32.attn.norm.scale": "model--00005-of-00007.safetensors",
|
||||
"block.32.attn.out.bias": "model--00005-of-00007.safetensors",
|
||||
"block.32.attn.out.weight": "model--00005-of-00007.safetensors",
|
||||
"block.32.attn.qkv.bias": "model--00005-of-00007.safetensors",
|
||||
"block.32.attn.qkv.weight": "model--00005-of-00007.safetensors",
|
||||
"block.32.attn.sinks": "model--00005-of-00007.safetensors",
|
||||
"block.32.mlp.gate.bias": "model--00005-of-00007.safetensors",
|
||||
"block.32.mlp.gate.weight": "model--00005-of-00007.safetensors",
|
||||
"block.32.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
|
||||
"block.32.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
|
||||
"block.32.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
|
||||
"block.32.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
|
||||
"block.32.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
|
||||
"block.32.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
|
||||
"block.32.mlp.norm.scale": "model--00005-of-00007.safetensors",
|
||||
"block.33.attn.norm.scale": "model--00005-of-00007.safetensors",
|
||||
"block.33.attn.out.bias": "model--00005-of-00007.safetensors",
|
||||
"block.33.attn.out.weight": "model--00005-of-00007.safetensors",
|
||||
"block.33.attn.qkv.bias": "model--00005-of-00007.safetensors",
|
||||
"block.33.attn.qkv.weight": "model--00005-of-00007.safetensors",
|
||||
"block.33.attn.sinks": "model--00005-of-00007.safetensors",
|
||||
"block.33.mlp.gate.bias": "model--00005-of-00007.safetensors",
|
||||
"block.33.mlp.gate.weight": "model--00005-of-00007.safetensors",
|
||||
"block.33.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
|
||||
"block.33.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
|
||||
"block.33.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
|
||||
"block.33.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
|
||||
"block.33.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
|
||||
"block.33.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
|
||||
"block.33.mlp.norm.scale": "model--00005-of-00007.safetensors",
|
||||
"block.34.attn.norm.scale": "model--00005-of-00007.safetensors",
|
||||
"block.34.attn.out.bias": "model--00005-of-00007.safetensors",
|
||||
"block.34.attn.out.weight": "model--00005-of-00007.safetensors",
|
||||
"block.34.attn.qkv.bias": "model--00005-of-00007.safetensors",
|
||||
"block.34.attn.qkv.weight": "model--00005-of-00007.safetensors",
|
||||
"block.34.attn.sinks": "model--00005-of-00007.safetensors",
|
||||
"block.34.mlp.gate.bias": "model--00005-of-00007.safetensors",
|
||||
"block.34.mlp.gate.weight": "model--00005-of-00007.safetensors",
|
||||
"block.34.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
|
||||
"block.34.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
|
||||
"block.34.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
|
||||
"block.34.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
|
||||
"block.34.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
|
||||
"block.34.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
|
||||
"block.34.mlp.norm.scale": "model--00005-of-00007.safetensors",
|
||||
"block.35.attn.norm.scale": "model--00005-of-00007.safetensors",
|
||||
"block.35.attn.out.bias": "model--00005-of-00007.safetensors",
|
||||
"block.35.attn.out.weight": "model--00005-of-00007.safetensors",
|
||||
"block.35.attn.qkv.bias": "model--00005-of-00007.safetensors",
|
||||
"block.35.attn.qkv.weight": "model--00005-of-00007.safetensors",
|
||||
"block.35.attn.sinks": "model--00005-of-00007.safetensors",
|
||||
"block.35.mlp.gate.bias": "model--00005-of-00007.safetensors",
|
||||
"block.35.mlp.gate.weight": "model--00005-of-00007.safetensors",
|
||||
"block.35.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
|
||||
"block.35.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
|
||||
"block.35.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
|
||||
"block.35.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
|
||||
"block.35.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
|
||||
"block.35.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
|
||||
"block.35.mlp.norm.scale": "model--00005-of-00007.safetensors",
|
||||
"block.4.attn.norm.scale": "model--00005-of-00007.safetensors",
|
||||
"block.4.attn.out.bias": "model--00005-of-00007.safetensors",
|
||||
"block.4.attn.out.weight": "model--00005-of-00007.safetensors",
|
||||
"block.4.attn.qkv.bias": "model--00005-of-00007.safetensors",
|
||||
"block.4.attn.qkv.weight": "model--00005-of-00007.safetensors",
|
||||
"block.4.attn.sinks": "model--00005-of-00007.safetensors",
|
||||
"block.4.mlp.gate.bias": "model--00005-of-00007.safetensors",
|
||||
"block.4.mlp.gate.weight": "model--00005-of-00007.safetensors",
|
||||
"block.4.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
|
||||
"block.4.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
|
||||
"block.4.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
|
||||
"block.4.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
|
||||
"block.4.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
|
||||
"block.4.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
|
||||
"block.4.mlp.norm.scale": "model--00006-of-00007.safetensors",
|
||||
"block.5.attn.norm.scale": "model--00006-of-00007.safetensors",
|
||||
"block.5.attn.out.bias": "model--00006-of-00007.safetensors",
|
||||
"block.5.attn.out.weight": "model--00006-of-00007.safetensors",
|
||||
"block.5.attn.qkv.bias": "model--00006-of-00007.safetensors",
|
||||
"block.5.attn.qkv.weight": "model--00006-of-00007.safetensors",
|
||||
"block.5.attn.sinks": "model--00006-of-00007.safetensors",
|
||||
"block.5.mlp.gate.bias": "model--00006-of-00007.safetensors",
|
||||
"block.5.mlp.gate.weight": "model--00006-of-00007.safetensors",
|
||||
"block.5.mlp.mlp1_bias": "model--00006-of-00007.safetensors",
|
||||
"block.5.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
|
||||
"block.5.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
|
||||
"block.5.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
|
||||
"block.5.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
|
||||
"block.5.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
|
||||
"block.5.mlp.norm.scale": "model--00006-of-00007.safetensors",
|
||||
"block.6.attn.norm.scale": "model--00006-of-00007.safetensors",
|
||||
"block.6.attn.out.bias": "model--00006-of-00007.safetensors",
|
||||
"block.6.attn.out.weight": "model--00006-of-00007.safetensors",
|
||||
"block.6.attn.qkv.bias": "model--00006-of-00007.safetensors",
|
||||
"block.6.attn.qkv.weight": "model--00006-of-00007.safetensors",
|
||||
"block.6.attn.sinks": "model--00006-of-00007.safetensors",
|
||||
"block.6.mlp.gate.bias": "model--00006-of-00007.safetensors",
|
||||
"block.6.mlp.gate.weight": "model--00006-of-00007.safetensors",
|
||||
"block.6.mlp.mlp1_bias": "model--00006-of-00007.safetensors",
|
||||
"block.6.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
|
||||
"block.6.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
|
||||
"block.6.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
|
||||
"block.6.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
|
||||
"block.6.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
|
||||
"block.6.mlp.norm.scale": "model--00006-of-00007.safetensors",
|
||||
"block.7.attn.norm.scale": "model--00006-of-00007.safetensors",
|
||||
"block.7.attn.out.bias": "model--00006-of-00007.safetensors",
|
||||
"block.7.attn.out.weight": "model--00006-of-00007.safetensors",
|
||||
"block.7.attn.qkv.bias": "model--00006-of-00007.safetensors",
|
||||
"block.7.attn.qkv.weight": "model--00006-of-00007.safetensors",
|
||||
"block.7.attn.sinks": "model--00006-of-00007.safetensors",
|
||||
"block.7.mlp.gate.bias": "model--00006-of-00007.safetensors",
|
||||
"block.7.mlp.gate.weight": "model--00006-of-00007.safetensors",
|
||||
"block.7.mlp.mlp1_bias": "model--00006-of-00007.safetensors",
|
||||
"block.7.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
|
||||
"block.7.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
|
||||
"block.7.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
|
||||
"block.7.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
|
||||
"block.7.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
|
||||
"block.7.mlp.norm.scale": "model--00006-of-00007.safetensors",
|
||||
"block.8.attn.norm.scale": "model--00006-of-00007.safetensors",
|
||||
"block.8.attn.out.bias": "model--00006-of-00007.safetensors",
|
||||
"block.8.attn.out.weight": "model--00006-of-00007.safetensors",
|
||||
"block.8.attn.qkv.bias": "model--00006-of-00007.safetensors",
|
||||
"block.8.attn.qkv.weight": "model--00006-of-00007.safetensors",
|
||||
"block.8.attn.sinks": "model--00006-of-00007.safetensors",
|
||||
"block.8.mlp.gate.bias": "model--00006-of-00007.safetensors",
|
||||
"block.8.mlp.gate.weight": "model--00006-of-00007.safetensors",
|
||||
"block.8.mlp.mlp1_bias": "model--00006-of-00007.safetensors",
|
||||
"block.8.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
|
||||
"block.8.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
|
||||
"block.8.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
|
||||
"block.8.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
|
||||
"block.8.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
|
||||
"block.8.mlp.norm.scale": "model--00006-of-00007.safetensors",
|
||||
"block.9.attn.norm.scale": "model--00006-of-00007.safetensors",
|
||||
"block.9.attn.out.bias": "model--00006-of-00007.safetensors",
|
||||
"block.9.attn.out.weight": "model--00006-of-00007.safetensors",
|
||||
"block.9.attn.qkv.bias": "model--00006-of-00007.safetensors",
|
||||
"block.9.attn.qkv.weight": "model--00006-of-00007.safetensors",
|
||||
"block.9.attn.sinks": "model--00006-of-00007.safetensors",
|
||||
"block.9.mlp.gate.bias": "model--00006-of-00007.safetensors",
|
||||
"block.9.mlp.gate.weight": "model--00006-of-00007.safetensors",
|
||||
"block.9.mlp.mlp1_bias": "model--00006-of-00007.safetensors",
|
||||
"block.9.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
|
||||
"block.9.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
|
||||
"block.9.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
|
||||
"block.9.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
|
||||
"block.9.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
|
||||
"block.9.mlp.norm.scale": "model--00006-of-00007.safetensors",
|
||||
"embedding.weight": "model--00007-of-00007.safetensors",
|
||||
"norm.scale": "model--00007-of-00007.safetensors",
|
||||
"unembedding.weight": "model--00007-of-00007.safetensors"
|
||||
}
|
||||
}
|
||||
Reference in New Issue
Block a user