Upload folder using huggingface_hub

Browse files

Files changed (11) hide show

config.json +3 -2
original/model--00001-of-00007.safetensors +0 -3
original/model--00002-of-00007.safetensors +0 -3
original/model--00003-of-00007.safetensors +0 -3
original/model--00004-of-00007.safetensors +0 -3
original/model--00005-of-00007.safetensors +0 -3
original/model--00006-of-00007.safetensors +0 -3
original/model--00007-of-00007.safetensors +0 -3
original/model.safetensors.index.json +0 -550
special_tokens_map.json +21 -3
tokenizer_config.json +7 -4

config.json CHANGED Viewed

@@ -58,7 +58,7 @@
   "num_key_value_heads": 8,
   "num_local_experts": 128,
   "output_router_logits": false,
-  "pad_token_id": 199999,
   "quantization_config": {
     "modules_to_not_convert": [
       "model.layers.*.self_attn",
@@ -82,7 +82,8 @@
   "sliding_window": 128,
   "swiglu_limit": 7.0,
   "tie_word_embeddings": false,
-  "transformers_version": "4.55.0.dev0",
   "use_cache": true,
   "vocab_size": 201088
 }

   "num_key_value_heads": 8,
   "num_local_experts": 128,
   "output_router_logits": false,
+  "pad_token_id": 200017,
   "quantization_config": {
     "modules_to_not_convert": [
       "model.layers.*.self_attn",
   "sliding_window": 128,
   "swiglu_limit": 7.0,
   "tie_word_embeddings": false,
+  "transformers_version": "4.56.0.dev0",
+  "unsloth_fixed": true,
   "use_cache": true,
   "vocab_size": 201088
 }

original/model--00001-of-00007.safetensors DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:68a8dc1f8e2e5996cb702f14332a25ddf3463daeab2df68e21ca09ef181203c3
-size 10544040680

original/model--00002-of-00007.safetensors DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:19b8f0d5c7dc3195c61a711d08384a1f85624f018186da541585c0f97ac61020
-size 10488721680

original/model--00003-of-00007.safetensors DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:0dbccd746d50e9543e8016d0a43ab4487c7f86d72349b1ef17abdfec509d0701
-size 10488721688

original/model--00004-of-00007.safetensors DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:bcc73cf6d18f96a2e62428758463157cc12768f410873152a50d3929a64cd049
-size 10488721672

original/model--00005-of-00007.safetensors DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:15fd69843e9cc6fdf2db0efe0cf0979b49a6ba84b3a38169b2fabc5479d04a7d
-size 10488721680

original/model--00006-of-00007.safetensors DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:3aedef2ee0a5a78a003b3f74fd6883033946b80097bf41e4f4715d95066f0588
-size 10433402600

original/model--00007-of-00007.safetensors DELETED Viewed

@@ -1,3 +0,0 @@
-version https://git-lfs.github.com/spec/v1
-oid sha256:20d5dfcad1ed6c50aa3c0da7d3f08828dba72b5f58686a987bf3a8f01659cda6
-size 2316539800

original/model.safetensors.index.json DELETED Viewed

@@ -1,550 +0,0 @@
-{
-  "metadata": {
-    "total_size": 65248815744
-  },
-  "weight_map": {
-    "block.0.attn.norm.scale": "model--00001-of-00007.safetensors",
-    "block.0.attn.out.bias": "model--00001-of-00007.safetensors",
-    "block.0.attn.out.weight": "model--00001-of-00007.safetensors",
-    "block.0.attn.qkv.bias": "model--00001-of-00007.safetensors",
-    "block.0.attn.qkv.weight": "model--00001-of-00007.safetensors",
-    "block.0.attn.sinks": "model--00001-of-00007.safetensors",
-    "block.0.mlp.gate.bias": "model--00001-of-00007.safetensors",
-    "block.0.mlp.gate.weight": "model--00001-of-00007.safetensors",
-    "block.0.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
-    "block.0.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
-    "block.0.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
-    "block.0.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
-    "block.0.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
-    "block.0.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
-    "block.0.mlp.norm.scale": "model--00001-of-00007.safetensors",
-    "block.1.attn.norm.scale": "model--00001-of-00007.safetensors",
-    "block.1.attn.out.bias": "model--00001-of-00007.safetensors",
-    "block.1.attn.out.weight": "model--00001-of-00007.safetensors",
-    "block.1.attn.qkv.bias": "model--00001-of-00007.safetensors",
-    "block.1.attn.qkv.weight": "model--00001-of-00007.safetensors",
-    "block.1.attn.sinks": "model--00001-of-00007.safetensors",
-    "block.1.mlp.gate.bias": "model--00001-of-00007.safetensors",
-    "block.1.mlp.gate.weight": "model--00001-of-00007.safetensors",
-    "block.1.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
-    "block.1.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
-    "block.1.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
-    "block.1.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
-    "block.1.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
-    "block.1.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
-    "block.1.mlp.norm.scale": "model--00001-of-00007.safetensors",
-    "block.10.attn.norm.scale": "model--00001-of-00007.safetensors",
-    "block.10.attn.out.bias": "model--00001-of-00007.safetensors",
-    "block.10.attn.out.weight": "model--00001-of-00007.safetensors",
-    "block.10.attn.qkv.bias": "model--00001-of-00007.safetensors",
-    "block.10.attn.qkv.weight": "model--00001-of-00007.safetensors",
-    "block.10.attn.sinks": "model--00001-of-00007.safetensors",
-    "block.10.mlp.gate.bias": "model--00001-of-00007.safetensors",
-    "block.10.mlp.gate.weight": "model--00001-of-00007.safetensors",
-    "block.10.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
-    "block.10.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
-    "block.10.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
-    "block.10.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
-    "block.10.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
-    "block.10.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
-    "block.10.mlp.norm.scale": "model--00001-of-00007.safetensors",
-    "block.11.attn.norm.scale": "model--00001-of-00007.safetensors",
-    "block.11.attn.out.bias": "model--00001-of-00007.safetensors",
-    "block.11.attn.out.weight": "model--00001-of-00007.safetensors",
-    "block.11.attn.qkv.bias": "model--00001-of-00007.safetensors",
-    "block.11.attn.qkv.weight": "model--00001-of-00007.safetensors",
-    "block.11.attn.sinks": "model--00001-of-00007.safetensors",
-    "block.11.mlp.gate.bias": "model--00001-of-00007.safetensors",
-    "block.11.mlp.gate.weight": "model--00001-of-00007.safetensors",
-    "block.11.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
-    "block.11.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
-    "block.11.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
-    "block.11.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
-    "block.11.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
-    "block.11.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
-    "block.11.mlp.norm.scale": "model--00001-of-00007.safetensors",
-    "block.12.attn.norm.scale": "model--00001-of-00007.safetensors",
-    "block.12.attn.out.bias": "model--00001-of-00007.safetensors",
-    "block.12.attn.out.weight": "model--00001-of-00007.safetensors",
-    "block.12.attn.qkv.bias": "model--00001-of-00007.safetensors",
-    "block.12.attn.qkv.weight": "model--00001-of-00007.safetensors",
-    "block.12.attn.sinks": "model--00001-of-00007.safetensors",
-    "block.12.mlp.gate.bias": "model--00001-of-00007.safetensors",
-    "block.12.mlp.gate.weight": "model--00001-of-00007.safetensors",
-    "block.12.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
-    "block.12.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
-    "block.12.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
-    "block.12.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
-    "block.12.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
-    "block.12.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
-    "block.12.mlp.norm.scale": "model--00001-of-00007.safetensors",
-    "block.13.attn.norm.scale": "model--00001-of-00007.safetensors",
-    "block.13.attn.out.bias": "model--00001-of-00007.safetensors",
-    "block.13.attn.out.weight": "model--00001-of-00007.safetensors",
-    "block.13.attn.qkv.bias": "model--00001-of-00007.safetensors",
-    "block.13.attn.qkv.weight": "model--00001-of-00007.safetensors",
-    "block.13.attn.sinks": "model--00001-of-00007.safetensors",
-    "block.13.mlp.gate.bias": "model--00001-of-00007.safetensors",
-    "block.13.mlp.gate.weight": "model--00001-of-00007.safetensors",
-    "block.13.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
-    "block.13.mlp.mlp1_weight.blocks": "model--00001-of-00007.safetensors",
-    "block.13.mlp.mlp1_weight.scales": "model--00001-of-00007.safetensors",
-    "block.13.mlp.mlp2_bias": "model--00001-of-00007.safetensors",
-    "block.13.mlp.mlp2_weight.blocks": "model--00001-of-00007.safetensors",
-    "block.13.mlp.mlp2_weight.scales": "model--00001-of-00007.safetensors",
-    "block.13.mlp.norm.scale": "model--00001-of-00007.safetensors",
-    "block.14.attn.norm.scale": "model--00001-of-00007.safetensors",
-    "block.14.attn.out.bias": "model--00001-of-00007.safetensors",
-    "block.14.attn.out.weight": "model--00001-of-00007.safetensors",
-    "block.14.attn.qkv.bias": "model--00001-of-00007.safetensors",
-    "block.14.attn.qkv.weight": "model--00001-of-00007.safetensors",
-    "block.14.attn.sinks": "model--00001-of-00007.safetensors",
-    "block.14.mlp.gate.bias": "model--00001-of-00007.safetensors",
-    "block.14.mlp.gate.weight": "model--00001-of-00007.safetensors",
-    "block.14.mlp.mlp1_bias": "model--00001-of-00007.safetensors",
-    "block.14.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
-    "block.14.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
-    "block.14.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
-    "block.14.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
-    "block.14.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
-    "block.14.mlp.norm.scale": "model--00002-of-00007.safetensors",
-    "block.15.attn.norm.scale": "model--00002-of-00007.safetensors",
-    "block.15.attn.out.bias": "model--00002-of-00007.safetensors",
-    "block.15.attn.out.weight": "model--00002-of-00007.safetensors",
-    "block.15.attn.qkv.bias": "model--00002-of-00007.safetensors",
-    "block.15.attn.qkv.weight": "model--00002-of-00007.safetensors",
-    "block.15.attn.sinks": "model--00002-of-00007.safetensors",
-    "block.15.mlp.gate.bias": "model--00002-of-00007.safetensors",
-    "block.15.mlp.gate.weight": "model--00002-of-00007.safetensors",
-    "block.15.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
-    "block.15.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
-    "block.15.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
-    "block.15.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
-    "block.15.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
-    "block.15.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
-    "block.15.mlp.norm.scale": "model--00002-of-00007.safetensors",
-    "block.16.attn.norm.scale": "model--00002-of-00007.safetensors",
-    "block.16.attn.out.bias": "model--00002-of-00007.safetensors",
-    "block.16.attn.out.weight": "model--00002-of-00007.safetensors",
-    "block.16.attn.qkv.bias": "model--00002-of-00007.safetensors",
-    "block.16.attn.qkv.weight": "model--00002-of-00007.safetensors",
-    "block.16.attn.sinks": "model--00002-of-00007.safetensors",
-    "block.16.mlp.gate.bias": "model--00002-of-00007.safetensors",
-    "block.16.mlp.gate.weight": "model--00002-of-00007.safetensors",
-    "block.16.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
-    "block.16.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
-    "block.16.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
-    "block.16.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
-    "block.16.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
-    "block.16.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
-    "block.16.mlp.norm.scale": "model--00002-of-00007.safetensors",
-    "block.17.attn.norm.scale": "model--00002-of-00007.safetensors",
-    "block.17.attn.out.bias": "model--00002-of-00007.safetensors",
-    "block.17.attn.out.weight": "model--00002-of-00007.safetensors",
-    "block.17.attn.qkv.bias": "model--00002-of-00007.safetensors",
-    "block.17.attn.qkv.weight": "model--00002-of-00007.safetensors",
-    "block.17.attn.sinks": "model--00002-of-00007.safetensors",
-    "block.17.mlp.gate.bias": "model--00002-of-00007.safetensors",
-    "block.17.mlp.gate.weight": "model--00002-of-00007.safetensors",
-    "block.17.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
-    "block.17.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
-    "block.17.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
-    "block.17.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
-    "block.17.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
-    "block.17.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
-    "block.17.mlp.norm.scale": "model--00002-of-00007.safetensors",
-    "block.18.attn.norm.scale": "model--00002-of-00007.safetensors",
-    "block.18.attn.out.bias": "model--00002-of-00007.safetensors",
-    "block.18.attn.out.weight": "model--00002-of-00007.safetensors",
-    "block.18.attn.qkv.bias": "model--00002-of-00007.safetensors",
-    "block.18.attn.qkv.weight": "model--00002-of-00007.safetensors",
-    "block.18.attn.sinks": "model--00002-of-00007.safetensors",
-    "block.18.mlp.gate.bias": "model--00002-of-00007.safetensors",
-    "block.18.mlp.gate.weight": "model--00002-of-00007.safetensors",
-    "block.18.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
-    "block.18.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
-    "block.18.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
-    "block.18.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
-    "block.18.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
-    "block.18.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
-    "block.18.mlp.norm.scale": "model--00002-of-00007.safetensors",
-    "block.19.attn.norm.scale": "model--00002-of-00007.safetensors",
-    "block.19.attn.out.bias": "model--00002-of-00007.safetensors",
-    "block.19.attn.out.weight": "model--00002-of-00007.safetensors",
-    "block.19.attn.qkv.bias": "model--00002-of-00007.safetensors",
-    "block.19.attn.qkv.weight": "model--00002-of-00007.safetensors",
-    "block.19.attn.sinks": "model--00002-of-00007.safetensors",
-    "block.19.mlp.gate.bias": "model--00002-of-00007.safetensors",
-    "block.19.mlp.gate.weight": "model--00002-of-00007.safetensors",
-    "block.19.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
-    "block.19.mlp.mlp1_weight.blocks": "model--00002-of-00007.safetensors",
-    "block.19.mlp.mlp1_weight.scales": "model--00002-of-00007.safetensors",
-    "block.19.mlp.mlp2_bias": "model--00002-of-00007.safetensors",
-    "block.19.mlp.mlp2_weight.blocks": "model--00002-of-00007.safetensors",
-    "block.19.mlp.mlp2_weight.scales": "model--00002-of-00007.safetensors",
-    "block.19.mlp.norm.scale": "model--00002-of-00007.safetensors",
-    "block.2.attn.norm.scale": "model--00002-of-00007.safetensors",
-    "block.2.attn.out.bias": "model--00002-of-00007.safetensors",
-    "block.2.attn.out.weight": "model--00002-of-00007.safetensors",
-    "block.2.attn.qkv.bias": "model--00002-of-00007.safetensors",
-    "block.2.attn.qkv.weight": "model--00002-of-00007.safetensors",
-    "block.2.attn.sinks": "model--00002-of-00007.safetensors",
-    "block.2.mlp.gate.bias": "model--00002-of-00007.safetensors",
-    "block.2.mlp.gate.weight": "model--00002-of-00007.safetensors",
-    "block.2.mlp.mlp1_bias": "model--00002-of-00007.safetensors",
-    "block.2.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
-    "block.2.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
-    "block.2.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
-    "block.2.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
-    "block.2.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
-    "block.2.mlp.norm.scale": "model--00003-of-00007.safetensors",
-    "block.20.attn.norm.scale": "model--00003-of-00007.safetensors",
-    "block.20.attn.out.bias": "model--00003-of-00007.safetensors",
-    "block.20.attn.out.weight": "model--00003-of-00007.safetensors",
-    "block.20.attn.qkv.bias": "model--00003-of-00007.safetensors",
-    "block.20.attn.qkv.weight": "model--00003-of-00007.safetensors",
-    "block.20.attn.sinks": "model--00003-of-00007.safetensors",
-    "block.20.mlp.gate.bias": "model--00003-of-00007.safetensors",
-    "block.20.mlp.gate.weight": "model--00003-of-00007.safetensors",
-    "block.20.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
-    "block.20.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
-    "block.20.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
-    "block.20.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
-    "block.20.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
-    "block.20.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
-    "block.20.mlp.norm.scale": "model--00003-of-00007.safetensors",
-    "block.21.attn.norm.scale": "model--00003-of-00007.safetensors",
-    "block.21.attn.out.bias": "model--00003-of-00007.safetensors",
-    "block.21.attn.out.weight": "model--00003-of-00007.safetensors",
-    "block.21.attn.qkv.bias": "model--00003-of-00007.safetensors",
-    "block.21.attn.qkv.weight": "model--00003-of-00007.safetensors",
-    "block.21.attn.sinks": "model--00003-of-00007.safetensors",
-    "block.21.mlp.gate.bias": "model--00003-of-00007.safetensors",
-    "block.21.mlp.gate.weight": "model--00003-of-00007.safetensors",
-    "block.21.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
-    "block.21.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
-    "block.21.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
-    "block.21.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
-    "block.21.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
-    "block.21.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
-    "block.21.mlp.norm.scale": "model--00003-of-00007.safetensors",
-    "block.22.attn.norm.scale": "model--00003-of-00007.safetensors",
-    "block.22.attn.out.bias": "model--00003-of-00007.safetensors",
-    "block.22.attn.out.weight": "model--00003-of-00007.safetensors",
-    "block.22.attn.qkv.bias": "model--00003-of-00007.safetensors",
-    "block.22.attn.qkv.weight": "model--00003-of-00007.safetensors",
-    "block.22.attn.sinks": "model--00003-of-00007.safetensors",
-    "block.22.mlp.gate.bias": "model--00003-of-00007.safetensors",
-    "block.22.mlp.gate.weight": "model--00003-of-00007.safetensors",
-    "block.22.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
-    "block.22.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
-    "block.22.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
-    "block.22.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
-    "block.22.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
-    "block.22.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
-    "block.22.mlp.norm.scale": "model--00003-of-00007.safetensors",
-    "block.23.attn.norm.scale": "model--00003-of-00007.safetensors",
-    "block.23.attn.out.bias": "model--00003-of-00007.safetensors",
-    "block.23.attn.out.weight": "model--00003-of-00007.safetensors",
-    "block.23.attn.qkv.bias": "model--00003-of-00007.safetensors",
-    "block.23.attn.qkv.weight": "model--00003-of-00007.safetensors",
-    "block.23.attn.sinks": "model--00003-of-00007.safetensors",
-    "block.23.mlp.gate.bias": "model--00003-of-00007.safetensors",
-    "block.23.mlp.gate.weight": "model--00003-of-00007.safetensors",
-    "block.23.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
-    "block.23.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
-    "block.23.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
-    "block.23.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
-    "block.23.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
-    "block.23.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
-    "block.23.mlp.norm.scale": "model--00003-of-00007.safetensors",
-    "block.24.attn.norm.scale": "model--00003-of-00007.safetensors",
-    "block.24.attn.out.bias": "model--00003-of-00007.safetensors",
-    "block.24.attn.out.weight": "model--00003-of-00007.safetensors",
-    "block.24.attn.qkv.bias": "model--00003-of-00007.safetensors",
-    "block.24.attn.qkv.weight": "model--00003-of-00007.safetensors",
-    "block.24.attn.sinks": "model--00003-of-00007.safetensors",
-    "block.24.mlp.gate.bias": "model--00003-of-00007.safetensors",
-    "block.24.mlp.gate.weight": "model--00003-of-00007.safetensors",
-    "block.24.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
-    "block.24.mlp.mlp1_weight.blocks": "model--00003-of-00007.safetensors",
-    "block.24.mlp.mlp1_weight.scales": "model--00003-of-00007.safetensors",
-    "block.24.mlp.mlp2_bias": "model--00003-of-00007.safetensors",
-    "block.24.mlp.mlp2_weight.blocks": "model--00003-of-00007.safetensors",
-    "block.24.mlp.mlp2_weight.scales": "model--00003-of-00007.safetensors",
-    "block.24.mlp.norm.scale": "model--00003-of-00007.safetensors",
-    "block.25.attn.norm.scale": "model--00003-of-00007.safetensors",
-    "block.25.attn.out.bias": "model--00003-of-00007.safetensors",
-    "block.25.attn.out.weight": "model--00003-of-00007.safetensors",
-    "block.25.attn.qkv.bias": "model--00003-of-00007.safetensors",
-    "block.25.attn.qkv.weight": "model--00003-of-00007.safetensors",
-    "block.25.attn.sinks": "model--00003-of-00007.safetensors",
-    "block.25.mlp.gate.bias": "model--00003-of-00007.safetensors",
-    "block.25.mlp.gate.weight": "model--00003-of-00007.safetensors",
-    "block.25.mlp.mlp1_bias": "model--00003-of-00007.safetensors",
-    "block.25.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
-    "block.25.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
-    "block.25.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
-    "block.25.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
-    "block.25.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
-    "block.25.mlp.norm.scale": "model--00004-of-00007.safetensors",
-    "block.26.attn.norm.scale": "model--00004-of-00007.safetensors",
-    "block.26.attn.out.bias": "model--00004-of-00007.safetensors",
-    "block.26.attn.out.weight": "model--00004-of-00007.safetensors",
-    "block.26.attn.qkv.bias": "model--00004-of-00007.safetensors",
-    "block.26.attn.qkv.weight": "model--00004-of-00007.safetensors",
-    "block.26.attn.sinks": "model--00004-of-00007.safetensors",
-    "block.26.mlp.gate.bias": "model--00004-of-00007.safetensors",
-    "block.26.mlp.gate.weight": "model--00004-of-00007.safetensors",
-    "block.26.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
-    "block.26.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
-    "block.26.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
-    "block.26.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
-    "block.26.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
-    "block.26.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
-    "block.26.mlp.norm.scale": "model--00004-of-00007.safetensors",
-    "block.27.attn.norm.scale": "model--00004-of-00007.safetensors",
-    "block.27.attn.out.bias": "model--00004-of-00007.safetensors",
-    "block.27.attn.out.weight": "model--00004-of-00007.safetensors",
-    "block.27.attn.qkv.bias": "model--00004-of-00007.safetensors",
-    "block.27.attn.qkv.weight": "model--00004-of-00007.safetensors",
-    "block.27.attn.sinks": "model--00004-of-00007.safetensors",
-    "block.27.mlp.gate.bias": "model--00004-of-00007.safetensors",
-    "block.27.mlp.gate.weight": "model--00004-of-00007.safetensors",
-    "block.27.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
-    "block.27.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
-    "block.27.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
-    "block.27.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
-    "block.27.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
-    "block.27.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
-    "block.27.mlp.norm.scale": "model--00004-of-00007.safetensors",
-    "block.28.attn.norm.scale": "model--00004-of-00007.safetensors",
-    "block.28.attn.out.bias": "model--00004-of-00007.safetensors",
-    "block.28.attn.out.weight": "model--00004-of-00007.safetensors",
-    "block.28.attn.qkv.bias": "model--00004-of-00007.safetensors",
-    "block.28.attn.qkv.weight": "model--00004-of-00007.safetensors",
-    "block.28.attn.sinks": "model--00004-of-00007.safetensors",
-    "block.28.mlp.gate.bias": "model--00004-of-00007.safetensors",
-    "block.28.mlp.gate.weight": "model--00004-of-00007.safetensors",
-    "block.28.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
-    "block.28.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
-    "block.28.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
-    "block.28.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
-    "block.28.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
-    "block.28.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
-    "block.28.mlp.norm.scale": "model--00004-of-00007.safetensors",
-    "block.29.attn.norm.scale": "model--00004-of-00007.safetensors",
-    "block.29.attn.out.bias": "model--00004-of-00007.safetensors",
-    "block.29.attn.out.weight": "model--00004-of-00007.safetensors",
-    "block.29.attn.qkv.bias": "model--00004-of-00007.safetensors",
-    "block.29.attn.qkv.weight": "model--00004-of-00007.safetensors",
-    "block.29.attn.sinks": "model--00004-of-00007.safetensors",
-    "block.29.mlp.gate.bias": "model--00004-of-00007.safetensors",
-    "block.29.mlp.gate.weight": "model--00004-of-00007.safetensors",
-    "block.29.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
-    "block.29.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
-    "block.29.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
-    "block.29.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
-    "block.29.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
-    "block.29.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
-    "block.29.mlp.norm.scale": "model--00004-of-00007.safetensors",
-    "block.3.attn.norm.scale": "model--00004-of-00007.safetensors",
-    "block.3.attn.out.bias": "model--00004-of-00007.safetensors",
-    "block.3.attn.out.weight": "model--00004-of-00007.safetensors",
-    "block.3.attn.qkv.bias": "model--00004-of-00007.safetensors",
-    "block.3.attn.qkv.weight": "model--00004-of-00007.safetensors",
-    "block.3.attn.sinks": "model--00004-of-00007.safetensors",
-    "block.3.mlp.gate.bias": "model--00004-of-00007.safetensors",
-    "block.3.mlp.gate.weight": "model--00004-of-00007.safetensors",
-    "block.3.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
-    "block.3.mlp.mlp1_weight.blocks": "model--00004-of-00007.safetensors",
-    "block.3.mlp.mlp1_weight.scales": "model--00004-of-00007.safetensors",
-    "block.3.mlp.mlp2_bias": "model--00004-of-00007.safetensors",
-    "block.3.mlp.mlp2_weight.blocks": "model--00004-of-00007.safetensors",
-    "block.3.mlp.mlp2_weight.scales": "model--00004-of-00007.safetensors",
-    "block.3.mlp.norm.scale": "model--00004-of-00007.safetensors",
-    "block.30.attn.norm.scale": "model--00004-of-00007.safetensors",
-    "block.30.attn.out.bias": "model--00004-of-00007.safetensors",
-    "block.30.attn.out.weight": "model--00004-of-00007.safetensors",
-    "block.30.attn.qkv.bias": "model--00004-of-00007.safetensors",
-    "block.30.attn.qkv.weight": "model--00004-of-00007.safetensors",
-    "block.30.attn.sinks": "model--00004-of-00007.safetensors",
-    "block.30.mlp.gate.bias": "model--00004-of-00007.safetensors",
-    "block.30.mlp.gate.weight": "model--00004-of-00007.safetensors",
-    "block.30.mlp.mlp1_bias": "model--00004-of-00007.safetensors",
-    "block.30.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
-    "block.30.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
-    "block.30.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
-    "block.30.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
-    "block.30.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
-    "block.30.mlp.norm.scale": "model--00005-of-00007.safetensors",
-    "block.31.attn.norm.scale": "model--00005-of-00007.safetensors",
-    "block.31.attn.out.bias": "model--00005-of-00007.safetensors",
-    "block.31.attn.out.weight": "model--00005-of-00007.safetensors",
-    "block.31.attn.qkv.bias": "model--00005-of-00007.safetensors",
-    "block.31.attn.qkv.weight": "model--00005-of-00007.safetensors",
-    "block.31.attn.sinks": "model--00005-of-00007.safetensors",
-    "block.31.mlp.gate.bias": "model--00005-of-00007.safetensors",
-    "block.31.mlp.gate.weight": "model--00005-of-00007.safetensors",
-    "block.31.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
-    "block.31.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
-    "block.31.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
-    "block.31.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
-    "block.31.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
-    "block.31.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
-    "block.31.mlp.norm.scale": "model--00005-of-00007.safetensors",
-    "block.32.attn.norm.scale": "model--00005-of-00007.safetensors",
-    "block.32.attn.out.bias": "model--00005-of-00007.safetensors",
-    "block.32.attn.out.weight": "model--00005-of-00007.safetensors",
-    "block.32.attn.qkv.bias": "model--00005-of-00007.safetensors",
-    "block.32.attn.qkv.weight": "model--00005-of-00007.safetensors",
-    "block.32.attn.sinks": "model--00005-of-00007.safetensors",
-    "block.32.mlp.gate.bias": "model--00005-of-00007.safetensors",
-    "block.32.mlp.gate.weight": "model--00005-of-00007.safetensors",
-    "block.32.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
-    "block.32.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
-    "block.32.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
-    "block.32.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
-    "block.32.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
-    "block.32.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
-    "block.32.mlp.norm.scale": "model--00005-of-00007.safetensors",
-    "block.33.attn.norm.scale": "model--00005-of-00007.safetensors",
-    "block.33.attn.out.bias": "model--00005-of-00007.safetensors",
-    "block.33.attn.out.weight": "model--00005-of-00007.safetensors",
-    "block.33.attn.qkv.bias": "model--00005-of-00007.safetensors",
-    "block.33.attn.qkv.weight": "model--00005-of-00007.safetensors",
-    "block.33.attn.sinks": "model--00005-of-00007.safetensors",
-    "block.33.mlp.gate.bias": "model--00005-of-00007.safetensors",
-    "block.33.mlp.gate.weight": "model--00005-of-00007.safetensors",
-    "block.33.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
-    "block.33.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
-    "block.33.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
-    "block.33.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
-    "block.33.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
-    "block.33.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
-    "block.33.mlp.norm.scale": "model--00005-of-00007.safetensors",
-    "block.34.attn.norm.scale": "model--00005-of-00007.safetensors",
-    "block.34.attn.out.bias": "model--00005-of-00007.safetensors",
-    "block.34.attn.out.weight": "model--00005-of-00007.safetensors",
-    "block.34.attn.qkv.bias": "model--00005-of-00007.safetensors",
-    "block.34.attn.qkv.weight": "model--00005-of-00007.safetensors",
-    "block.34.attn.sinks": "model--00005-of-00007.safetensors",
-    "block.34.mlp.gate.bias": "model--00005-of-00007.safetensors",
-    "block.34.mlp.gate.weight": "model--00005-of-00007.safetensors",
-    "block.34.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
-    "block.34.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
-    "block.34.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
-    "block.34.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
-    "block.34.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
-    "block.34.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
-    "block.34.mlp.norm.scale": "model--00005-of-00007.safetensors",
-    "block.35.attn.norm.scale": "model--00005-of-00007.safetensors",
-    "block.35.attn.out.bias": "model--00005-of-00007.safetensors",
-    "block.35.attn.out.weight": "model--00005-of-00007.safetensors",
-    "block.35.attn.qkv.bias": "model--00005-of-00007.safetensors",
-    "block.35.attn.qkv.weight": "model--00005-of-00007.safetensors",
-    "block.35.attn.sinks": "model--00005-of-00007.safetensors",
-    "block.35.mlp.gate.bias": "model--00005-of-00007.safetensors",
-    "block.35.mlp.gate.weight": "model--00005-of-00007.safetensors",
-    "block.35.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
-    "block.35.mlp.mlp1_weight.blocks": "model--00005-of-00007.safetensors",
-    "block.35.mlp.mlp1_weight.scales": "model--00005-of-00007.safetensors",
-    "block.35.mlp.mlp2_bias": "model--00005-of-00007.safetensors",
-    "block.35.mlp.mlp2_weight.blocks": "model--00005-of-00007.safetensors",
-    "block.35.mlp.mlp2_weight.scales": "model--00005-of-00007.safetensors",
-    "block.35.mlp.norm.scale": "model--00005-of-00007.safetensors",
-    "block.4.attn.norm.scale": "model--00005-of-00007.safetensors",
-    "block.4.attn.out.bias": "model--00005-of-00007.safetensors",
-    "block.4.attn.out.weight": "model--00005-of-00007.safetensors",
-    "block.4.attn.qkv.bias": "model--00005-of-00007.safetensors",
-    "block.4.attn.qkv.weight": "model--00005-of-00007.safetensors",
-    "block.4.attn.sinks": "model--00005-of-00007.safetensors",
-    "block.4.mlp.gate.bias": "model--00005-of-00007.safetensors",
-    "block.4.mlp.gate.weight": "model--00005-of-00007.safetensors",
-    "block.4.mlp.mlp1_bias": "model--00005-of-00007.safetensors",
-    "block.4.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
-    "block.4.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
-    "block.4.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
-    "block.4.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
-    "block.4.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
-    "block.4.mlp.norm.scale": "model--00006-of-00007.safetensors",
-    "block.5.attn.norm.scale": "model--00006-of-00007.safetensors",
-    "block.5.attn.out.bias": "model--00006-of-00007.safetensors",
-    "block.5.attn.out.weight": "model--00006-of-00007.safetensors",
-    "block.5.attn.qkv.bias": "model--00006-of-00007.safetensors",
-    "block.5.attn.qkv.weight": "model--00006-of-00007.safetensors",
-    "block.5.attn.sinks": "model--00006-of-00007.safetensors",
-    "block.5.mlp.gate.bias": "model--00006-of-00007.safetensors",
-    "block.5.mlp.gate.weight": "model--00006-of-00007.safetensors",
-    "block.5.mlp.mlp1_bias": "model--00006-of-00007.safetensors",
-    "block.5.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
-    "block.5.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
-    "block.5.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
-    "block.5.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
-    "block.5.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
-    "block.5.mlp.norm.scale": "model--00006-of-00007.safetensors",
-    "block.6.attn.norm.scale": "model--00006-of-00007.safetensors",
-    "block.6.attn.out.bias": "model--00006-of-00007.safetensors",
-    "block.6.attn.out.weight": "model--00006-of-00007.safetensors",
-    "block.6.attn.qkv.bias": "model--00006-of-00007.safetensors",
-    "block.6.attn.qkv.weight": "model--00006-of-00007.safetensors",
-    "block.6.attn.sinks": "model--00006-of-00007.safetensors",
-    "block.6.mlp.gate.bias": "model--00006-of-00007.safetensors",
-    "block.6.mlp.gate.weight": "model--00006-of-00007.safetensors",
-    "block.6.mlp.mlp1_bias": "model--00006-of-00007.safetensors",
-    "block.6.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
-    "block.6.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
-    "block.6.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
-    "block.6.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
-    "block.6.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
-    "block.6.mlp.norm.scale": "model--00006-of-00007.safetensors",
-    "block.7.attn.norm.scale": "model--00006-of-00007.safetensors",
-    "block.7.attn.out.bias": "model--00006-of-00007.safetensors",
-    "block.7.attn.out.weight": "model--00006-of-00007.safetensors",
-    "block.7.attn.qkv.bias": "model--00006-of-00007.safetensors",
-    "block.7.attn.qkv.weight": "model--00006-of-00007.safetensors",
-    "block.7.attn.sinks": "model--00006-of-00007.safetensors",
-    "block.7.mlp.gate.bias": "model--00006-of-00007.safetensors",
-    "block.7.mlp.gate.weight": "model--00006-of-00007.safetensors",
-    "block.7.mlp.mlp1_bias": "model--00006-of-00007.safetensors",
-    "block.7.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
-    "block.7.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
-    "block.7.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
-    "block.7.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
-    "block.7.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
-    "block.7.mlp.norm.scale": "model--00006-of-00007.safetensors",
-    "block.8.attn.norm.scale": "model--00006-of-00007.safetensors",
-    "block.8.attn.out.bias": "model--00006-of-00007.safetensors",
-    "block.8.attn.out.weight": "model--00006-of-00007.safetensors",
-    "block.8.attn.qkv.bias": "model--00006-of-00007.safetensors",
-    "block.8.attn.qkv.weight": "model--00006-of-00007.safetensors",
-    "block.8.attn.sinks": "model--00006-of-00007.safetensors",
-    "block.8.mlp.gate.bias": "model--00006-of-00007.safetensors",
-    "block.8.mlp.gate.weight": "model--00006-of-00007.safetensors",
-    "block.8.mlp.mlp1_bias": "model--00006-of-00007.safetensors",
-    "block.8.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
-    "block.8.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
-    "block.8.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
-    "block.8.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
-    "block.8.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
-    "block.8.mlp.norm.scale": "model--00006-of-00007.safetensors",
-    "block.9.attn.norm.scale": "model--00006-of-00007.safetensors",
-    "block.9.attn.out.bias": "model--00006-of-00007.safetensors",
-    "block.9.attn.out.weight": "model--00006-of-00007.safetensors",
-    "block.9.attn.qkv.bias": "model--00006-of-00007.safetensors",
-    "block.9.attn.qkv.weight": "model--00006-of-00007.safetensors",
-    "block.9.attn.sinks": "model--00006-of-00007.safetensors",
-    "block.9.mlp.gate.bias": "model--00006-of-00007.safetensors",
-    "block.9.mlp.gate.weight": "model--00006-of-00007.safetensors",
-    "block.9.mlp.mlp1_bias": "model--00006-of-00007.safetensors",
-    "block.9.mlp.mlp1_weight.blocks": "model--00006-of-00007.safetensors",
-    "block.9.mlp.mlp1_weight.scales": "model--00006-of-00007.safetensors",
-    "block.9.mlp.mlp2_bias": "model--00006-of-00007.safetensors",
-    "block.9.mlp.mlp2_weight.blocks": "model--00006-of-00007.safetensors",
-    "block.9.mlp.mlp2_weight.scales": "model--00006-of-00007.safetensors",
-    "block.9.mlp.norm.scale": "model--00006-of-00007.safetensors",
-    "embedding.weight": "model--00007-of-00007.safetensors",
-    "norm.scale": "model--00007-of-00007.safetensors",
-    "unembedding.weight": "model--00007-of-00007.safetensors"
-  }
-}

special_tokens_map.json CHANGED Viewed

@@ -1,5 +1,23 @@
 {
-  "bos_token": "<|startoftext|>",
-  "eos_token": "<|return|>",
-  "pad_token": "<|endoftext|>"
 }

 {
+  "bos_token": {
+    "content": "<|startoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "<|return|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<|reserved_200017|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
 }

tokenizer_config.json CHANGED Viewed

@@ -177,7 +177,10 @@
     "input_ids",
     "attention_mask"
   ],
-  "model_max_length": 1000000000000000019884624838656,
-  "pad_token": "<|endoftext|>",
-  "tokenizer_class": "PreTrainedTokenizerFast"
-}

     "input_ids",
     "attention_mask"
   ],
+  "model_max_length": 131072,
+  "pad_token": "<|reserved_200017|>",
+  "padding_side": "left",
+  "tokenizer_class": "PreTrainedTokenizerFast",
+  "unk_token": null,
+  "chat_template": "{# Copyright 2025-present Unsloth. Apache 2.0 License. Unsloth chat template fixes. Edited from ggml-org & OpenAI #}\n{#-\n  In addition to the normal inputs of `messages` and `tools`, this template also accepts the\n  following kwargs:\n  - \"builtin_tools\": A list, can contain \"browser\" and/or \"python\".\n  - \"model_identity\": A string that optionally describes the model identity.\n  - \"reasoning_effort\": A string that describes the reasoning effort, defaults to \"medium\".\n #}\n\n{#- Tool Definition Rendering ============================================== #}\n{%- macro render_typescript_type(param_spec, required_params, is_nullable=false) -%}\n    {%- if param_spec.type == \"array\" -%}\n        {%- if param_spec['items'] -%}\n            {%- if param_spec['items']['type'] == \"string\" -%}\n                {{- \"string[]\" }}\n            {%- elif param_spec['items']['type'] == \"number\" -%}\n                {{- \"number[]\" }}\n            {%- elif param_spec['items']['type'] == \"integer\" -%}\n                {{- \"number[]\" }}\n            {%- elif param_spec['items']['type'] == \"boolean\" -%}\n                {{- \"boolean[]\" }}\n            {%- else -%}\n                {%- set inner_type = render_typescript_type(param_spec['items'], required_params) -%}\n                {%- if inner_type == \"object | object\" or inner_type|length > 50 -%}\n                    {{- \"any[]\" }}\n                {%- else -%}\n                    {{- inner_type + \"[]\" }}\n                {%- endif -%}\n            {%- endif -%}\n            {%- if param_spec.nullable -%}\n                {{- \" | null\" }}\n            {%- endif -%}\n        {%- else -%}\n            {{- \"any[]\" }}\n            {%- if param_spec.nullable -%}\n                {{- \" | null\" }}\n            {%- endif -%}\n        {%- endif -%}\n    {%- elif param_spec.type is defined and param_spec.type is iterable and param_spec.type is not string and param_spec.type is not mapping and param_spec.type[0] is defined -%}\n        {#- Handle array of types like [\"object\", \"object\"] from Union[dict, list] #}\n        {%- if param_spec.type | length > 1 -%}\n            {{- param_spec.type | join(\" | \") }}\n        {%- else -%}\n            {{- param_spec.type[0] }}\n        {%- endif -%}\n    {%- elif param_spec.oneOf -%}\n        {#- Handle oneOf schemas - check for complex unions and fallback to any #}\n        {%- set has_object_variants = false -%}\n        {%- for variant in param_spec.oneOf -%}\n            {%- if variant.type == \"object\" -%}\n                {%- set has_object_variants = true -%}\n            {%- endif -%}\n        {%- endfor -%}\n        {%- if has_object_variants and param_spec.oneOf|length > 1 -%}\n            {{- \"any\" }}\n        {%- else -%}\n            {%- for variant in param_spec.oneOf -%}\n                {{- render_typescript_type(variant, required_params) -}}\n                {%- if variant.description %}\n                    {{- \"// \" + variant.description }}\n                {%- endif -%}\n                {%- if variant.default is defined %}\n                    {{ \"// default: \" + variant.default|tojson }}\n                {%- endif -%}\n                {%- if not loop.last %}\n                    {{- \" | \" }}\n                {% endif -%}\n            {%- endfor -%}\n        {%- endif -%}\n    {%- elif param_spec.type == \"string\" -%}\n        {%- if param_spec.enum -%}\n            {{- '\"' + param_spec.enum|join('\" | \"') + '\"' -}}\n        {%- else -%}\n            {{- \"string\" }}\n            {%- if param_spec.nullable %}\n                {{- \" | null\" }}\n            {%- endif -%}\n        {%- endif -%}\n    {%- elif param_spec.type == \"number\" -%}\n        {{- \"number\" }}\n    {%- elif param_spec.type == \"integer\" -%}\n        {{- \"number\" }}\n    {%- elif param_spec.type == \"boolean\" -%}\n        {{- \"boolean\" }}\n\n    {%- elif param_spec.type == \"object\" -%}\n        {%- if param_spec.properties -%}\n            {{- \"{\\n\" }}\n            {%- for prop_name, prop_spec in param_spec.properties.items() -%}\n                {{- prop_name -}}\n                {%- if prop_name not in (param_spec.required or []) -%}\n                    {{- \"?\" }}\n                {%- endif -%}\n                {{- \": \" }}\n                {{ render_typescript_type(prop_spec, param_spec.required or []) }}\n                {%- if not loop.last -%}\n                    {{-\", \" }}\n                {%- endif -%}\n            {%- endfor -%}\n            {{- \"}\" }}\n        {%- else -%}\n            {{- \"object\" }}\n        {%- endif -%}\n    {%- else -%}\n        {{- \"any\" }}\n    {%- endif -%}\n{%- endmacro -%}\n\n{%- macro render_tool_namespace(namespace_name, tools) -%}\n    {{- \"## \" + namespace_name + \"\\n\\n\" }}\n    {{- \"namespace \" + namespace_name + \" {\\n\\n\" }}\n    {%- for tool in tools %}\n        {%- set tool = tool.function %}\n        {{- \"// \" + tool.description + \"\\n\" }}\n        {{- \"type \"+ tool.name + \" = \" }}\n        {%- if tool.parameters and tool.parameters.properties -%}\n            {{- \"(_: \" }}\n            {{- \"{\\n\" }}\n            {%- for param_name, param_spec in tool.parameters.properties.items() %}\n                {{- \"// \" + param_spec.description + \"\\n\" }}\n                {{- param_name }}\n                {%- if param_name not in (tool.parameters.required or []) -%}\n                    {{- \"?\" }}\n                {%- endif -%}\n                {{- \": \" }}\n                {{- render_typescript_type(param_spec, tool.parameters.required or []) }}\n                {%- if param_spec.default is defined -%}\n                    {%- if param_spec.enum %}\n                        {{- \", // default: \" + param_spec.default }}\n                    {%- elif param_spec.oneOf %}\n                        {{- \"// default: \" + param_spec.default }}\n                    {%- else %}\n                        {{- \", // default: \" + param_spec.default|tojson }}\n                    {%- endif -%}\n                {%- endif -%}\n                {%- if not loop.last %}\n                    {{- \",\\n\" }}\n                {%- else %}\n                    {{- \"\\n\" }}\n                {%- endif -%}\n            {%- endfor %}\n            {{- \"}) => any;\\n\\n\" }}\n        {%- else -%}\n            {{- \"() => any;\\n\\n\" }}\n        {%- endif -%}\n    {%- endfor %}\n    {{- \"} // namespace \" + namespace_name }}\n{%- endmacro -%}\n\n{%- macro render_builtin_tools(browser_tool, python_tool) -%}\n    {%- if browser_tool %}\n        {{- \"## browser\\n\\n\" }}\n        {{- \"// Tool for browsing.\\n\" }}\n        {{- \"// The `cursor` appears in brackets before each browsing display: `[{cursor}]`.\\n\" }}\n        {{- \"// Cite information from the tool using the following format:\\n\" }}\n        {{- \"// `【{cursor}†L{line_start}(-L{line_end})?】`, for example: `【6†L9-L11】` or `【8†L3】`.\\n\" }}\n        {{- \"// Do not quote more than 10 words directly from the tool output.\\n\" }}\n        {{- \"// sources=web (default: web)\\n\" }}\n        {{- \"namespace browser {\\n\\n\" }}\n        {{- \"// Searches for information related to `query` and displays `topn` results.\\n\" }}\n        {{- \"type search = (_: {\\n\" }}\n        {{- \"query: string,\\n\" }}\n        {{- \"topn?: number, // default: 10\\n\" }}\n        {{- \"source?: string,\\n\" }}\n        {{- \"}) => any;\\n\\n\" }}\n        {{- \"// Opens the link `id` from the page indicated by `cursor` starting at line number `loc`, showing `num_lines` lines.\\n\" }}\n        {{- \"// Valid link ids are displayed with the formatting: `【{id}†.*】`.\\n\" }}\n        {{- \"// If `cursor` is not provided, the most recent page is implied.\\n\" }}\n        {{- \"// If `id` is a string, it is treated as a fully qualified URL associated with `source`.\\n\" }}\n        {{- \"// If `loc` is not provided, the viewport will be positioned at the beginning of the document or centered on the most relevant passage, if available.\\n\" }}\n        {{- \"// Use this function without `id` to scroll to a new location of an opened page.\\n\" }}\n        {{- \"type open = (_: {\\n\" }}\n        {{- \"id?: number | string, // default: -1\\n\" }}\n        {{- \"cursor?: number, // default: -1\\n\" }}\n        {{- \"loc?: number, // default: -1\\n\" }}\n        {{- \"num_lines?: number, // default: -1\\n\" }}\n        {{- \"view_source?: boolean, // default: false\\n\" }}\n        {{- \"source?: string,\\n\" }}\n        {{- \"}) => any;\\n\\n\" }}\n        {{- \"// Finds exact matches of `pattern` in the current page, or the page given by `cursor`.\\n\" }}\n        {{- \"type find = (_: {\\n\" }}\n        {{- \"pattern: string,\\n\" }}\n        {{- \"cursor?: number, // default: -1\\n\" }}\n        {{- \"}) => any;\\n\\n\" }}\n        {{- \"} // namespace browser\\n\\n\" }}\n    {%- endif -%}\n\n    {%- if python_tool %}\n        {{- \"## python\\n\\n\" }}\n        {{- \"Use this tool to execute Python code in your chain of thought. The code will not be shown to the user. This tool should be used for internal reasoning, but not for code that is intended to be visible to the user (e.g. when creating plots, tables, or files).\\n\\n\" }}\n        {{- \"When you send a message containing Python code to python, it will be executed in a stateful Jupyter notebook environment. python will respond with the output of the execution or time out after 120.0 seconds. The drive at '/mnt/data' can be used to save and persist user files. Internet access for this session is UNKNOWN. Depends on the cluster.\\n\\n\" }}\n    {%- endif -%}\n{%- endmacro -%}\n\n{#- System Message Construction ============================================ #}\n{%- macro build_system_message() -%}\n    {%- if model_identity is not defined %}\n        {{- \"You are ChatGPT, a large language model trained by OpenAI.\\n\" -}}\n    {%- else %}\n        {{- model_identity }}\n    {%- endif %}\n    {{- \"Knowledge cutoff: 2024-06\\n\" }}\n    {{- \"Current date: \" + strftime_now(\"%Y-%m-%d\") + \"\\n\\n\" }}\n    {%- if reasoning_effort is not defined %}\n        {%- set reasoning_effort = \"medium\" %}\n    {%- endif %}\n    {{- \"Reasoning: \" + reasoning_effort + \"\\n\\n\" }}\n    {%- if builtin_tools is defined %}\n        {{- \"# Tools\\n\\n\" }}\n        {%- set available_builtin_tools = namespace(browser=false, python=false) %}\n        {%- for tool in builtin_tools %}\n            {%- if tool == \"browser\" %}\n                {%- set available_builtin_tools.browser = true %}\n            {%- elif tool == \"python\" %}\n                {%- set available_builtin_tools.python = true %}\n            {%- endif %}\n        {%- endfor %}\n        {{- render_builtin_tools(available_builtin_tools.browser, available_builtin_tools.python) }}\n    {%- endif -%}\n    {{- \"# Valid channels: analysis, commentary, final. Channel must be included for every message.\" }}\n    {%- if tools is defined -%}\n        {{- \"\\nCalls to these tools must go to the commentary channel: 'functions'.\" }}\n    {%- endif -%}\n{%- endmacro -%}\n\n{#- Main Template Logic ================================================= #}\n{#- Set defaults #}\n\n{#- Render system message #}\n{{- \"<|start|>system<|message|>\" }}\n{{- build_system_message() }}\n{{- \"<|end|>\" }}\n\n{#- Extract developer message #}\n{%- if messages[0].role == \"developer\" or messages[0].role == \"system\" %}\n    {%- set developer_message = messages[0].content %}\n    {%- set loop_messages = messages[1:] %}\n{%- else %}\n    {%- set developer_message = \"\" %}\n    {%- set loop_messages = messages %}\n{%- endif %}\n\n{#- Render developer message #}\n{%- if developer_message or tools %}\n    {{- \"<|start|>developer<|message|>\" }}\n    {%- if developer_message %}\n        {{- \"# Instructions\\n\\n\" }}\n        {{- developer_message }}\n    {%- endif %}\n    {%- if tools -%}\n        {{- \"\\n\\n\" }}\n        {{- \"# Tools\\n\\n\" }}\n        {{- render_tool_namespace(\"functions\", tools) }}\n    {%- endif -%}\n    {{- \"<|end|>\" }}\n{%- endif %}\n\n{#- Render messages #}\n{%- set last_tool_call = namespace(name=none) %}\n{%- for message in loop_messages -%}\n    {#- At this point only assistant/user/tool messages should remain #}\n    {%- if message.role == 'assistant' -%}\n        {%- if \"tool_calls\" in message %}\n            {#- We assume max 1 tool call per message, and so we infer the tool call name #}\n            {#- in \"tool\" messages from the most recent assistant tool call name #}\n            {%- set tool_call = message.tool_calls[0] %}\n            {%- if tool_call.function %}\n                {%- set tool_call = tool_call.function %}\n            {%- endif %}\n            {%- if message.content %}\n                {{- \"<|start|>assistant<|channel|>analysis<|message|>\" + message.content + \"<|end|>\" }}\n            {%- endif %}\n            {{- \"<|start|>assistant to=\" }}\n            {{- \"functions.\" + tool_call.name + \"<|channel|>commentary json<|message|>\" }}\n            {{- tool_call.arguments|tojson }}\n            {{- \"<|call|>\" }}\n            {%- set last_tool_call.name = tool_call.name %}\n        {%- elif \"thinking\" in message and loop.last and not add_generation_prompt %}\n            {#- Only render the CoT if the final turn is an assistant turn and add_generation_prompt is false #}\n            {#- This is a situation that should only occur in training, never in inference. #}\n            {{- \"<|start|>assistant<|channel|>analysis<|message|>\" + message.thinking + \"<|end|>\" }}\n            {#- <|return|> indicates the end of generation, but <|end|> does not #}\n            {#- <|return|> should never be an input to the model, but we include it as the final token #}\n            {#- when training, so the model learns to emit it. #}\n            {{- \"<|start|>assistant<|channel|>final<|message|>\" + message.content + \"<|return|>\" }}\n            {%- set last_tool_call.name = none %}\n        {%- elif \"thinking\" in message %}\n            {#- CoT is dropped during all previous turns, so we never render it for inference #}\n            {{- \"<|start|>assistant<|channel|>final<|message|>\" + message.content + \"<|end|>\" }}\n            {%- set last_tool_call.name = none %}\n        {%- elif loop.last and not add_generation_prompt %}\n            {#- <|return|> indicates the end of generation, but <|end|> does not #}\n            {#- <|return|> should never be an input to the model, but we include it as the final token #}\n            {#- when training, so the model learns to emit it. #}\n            {{- \"<|start|>assistant<|message|>\" + message.content + \"<|return|>\" }}\n        {%- else %}\n            {{- \"<|start|>assistant<|message|>\" + message.content + \"<|end|>\" }}\n            {%- set last_tool_call.name = none %}\n        {%- endif %}\n    {%- elif message.role == 'tool' -%}\n        {%- if last_tool_call.name is none %}\n            {{- raise_exception(\"Message has tool role, but there was no previous assistant message with a tool call!\") }}\n        {%- endif %}\n        {{- \"<|start|>functions.\" + last_tool_call.name }}\n        {{- \" to=assistant<|channel|>commentary<|message|>\" + message.content|tojson + \"<|end|>\" }}\n    {%- else -%}\n        {{- \"<|start|>user<|message|>\" + message.content + \"<|end|>\" }}\n    {%- endif -%}\n{%- endfor -%}\n\n{#- Generation prompt #}\n{%- if add_generation_prompt -%}\n<|start|>assistant\n{%- endif -%}\n{# Copyright 2025-present Unsloth. Apache 2.0 License. Unsloth chat template fixes. Edited from ggml-org & OpenAI #}"
+}