Model save

Browse files

Files changed (12) hide show

README.md +65 -0
config.json +96 -0
events.out.tfevents.1741447692.ef0fed4470ac.1412.0manual_logs +3 -0
events.out.tfevents.1741447693.ef0fed4470ac.1412.1 +3 -0
events.out.tfevents.1741450037.ef0fed4470ac.1412.2 +3 -0
merges.txt +0 -0
model.safetensors +3 -0
special_tokens_map.json +34 -0
tokenizer.json +0 -0
tokenizer_config.json +155 -0
training_args.bin +3 -0
vocab.json +0 -0

README.md ADDED Viewed

	@@ -0,0 +1,65 @@

+---
+library_name: transformers
+tags:
+- generated_from_trainer
+model-index:
+- name: SlimPajama_10k_SmolLM2-360M_distil_ratio_no_additive_norm
+  results: []
+---
+<!-- This model card has been generated automatically according to the information the Trainer had access to. You
+should probably proofread and complete it, then remove this comment. -->
+# SlimPajama_10k_SmolLM2-360M_distil_ratio_no_additive_norm
+This model is a fine-tuned version of [](https://huggingface.co/) on the None dataset.
+It achieves the following results on the evaluation set:
+- Loss: 5.9171
+## Model description
+More information needed
+## Intended uses & limitations
+More information needed
+## Training and evaluation data
+More information needed
+## Training procedure
+### Training hyperparameters
+The following hyperparameters were used during training:
+- learning_rate: 0.0002
+- train_batch_size: 5
+- eval_batch_size: 5
+- seed: 42
+- optimizer: Use adamw_torch_fused with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments
+- lr_scheduler_type: cosine
+- lr_scheduler_warmup_ratio: 0.1
+- num_epochs: 1
+- mixed_precision_training: Native AMP
+### Training results
+| Training Loss | Epoch | Step | Validation Loss |
+|:-------------:|:-----:|:----:|:---------------:|
+| 9.1473        | 0.125 | 250  | 7.2546          |
+| 7.0097        | 0.25  | 500  | 6.7396          |
+| 6.6398        | 0.375 | 750  | 6.4495          |
+| 6.4032        | 0.5   | 1000 | 6.2413          |
+| 6.204         | 0.625 | 1250 | 6.0940          |
+| 6.07          | 0.75  | 1500 | 5.9828          |
+| 5.9685        | 0.875 | 1750 | 5.9299          |
+| 5.9672        | 1.0   | 2000 | 5.9171          |
+### Framework versions
+- Transformers 4.48.3
+- Pytorch 2.5.1+cu124
+- Datasets 3.3.2
+- Tokenizers 0.21.0

config.json ADDED Viewed

	@@ -0,0 +1,96 @@

+{
+  "architectures": [
+    "DistilxLSTM"
+  ],
+  "model_type": "xlstm",
+  "pad_token_id": 2,
+  "torch_dtype": "float32",
+  "transformers_version": "4.48.3",
+  "xlstm_cfg": {
+    "_block_map": "1,0,0,0,1,0,0,0,1,0,0,0,1,0,0,0",
+    "add_embedding_dropout": false,
+    "add_post_blocks_norm": true,
+    "bias": false,
+    "context_length": 256,
+    "dropout": 0.0,
+    "embedding_dim": 960,
+    "mlstm_block": {
+      "_block_idx": null,
+      "_num_blocks": 16,
+      "mlstm": {
+        "_inner_embedding_dim": 1920,
+        "_num_blocks": 16,
+        "_proj_up_dim": 1920,
+        "bias": false,
+        "context_length": 256,
+        "conv1d_kernel_size": 4,
+        "dropout": 0.0,
+        "embedding_dim": 960,
+        "num_heads": 4,
+        "proj_factor": 2.0,
+        "qkv_proj_blocksize": 32,
+        "round_proj_up_dim_up": true,
+        "round_proj_up_to_multiple_of": 64
+      }
+    },
+    "num_blocks": 16,
+    "slstm_at": [
+      0,
+      4,
+      8,
+      12
+    ],
+    "slstm_block": {
+      "_block_idx": null,
+      "_num_blocks": 16,
+      "feedforward": {
+        "_num_blocks": 1,
+        "_proj_up_dim": 0,
+        "act_fn": "gelu",
+        "bias": false,
+        "dropout": 0.0,
+        "embedding_dim": -1,
+        "ff_type": "ffn_gated",
+        "proj_factor": 1.7,
+        "round_proj_up_dim_up": true,
+        "round_proj_up_to_multiple_of": 64
+      },
+      "slstm": {
+        "_block_idx": null,
+        "_num_blocks": 16,
+        "backend": "cuda",
+        "batch_size": 8,
+        "bias_init": "powerlaw_blockdependent",
+        "constants": {},
+        "conv1d_kernel_size": 4,
+        "dropout": 0.0,
+        "dtype": "bfloat16",
+        "dtype_a": "float32",
+        "dtype_b": "float32",
+        "dtype_g": "bfloat16",
+        "dtype_r": "bfloat16",
+        "dtype_s": "bfloat16",
+        "dtype_w": "bfloat16",
+        "embedding_dim": 960,
+        "enable_automatic_mixed_precision": true,
+        "forward_clipval": null,
+        "function": "slstm",
+        "gradient_recurrent_clipval": null,
+        "gradient_recurrent_cut": false,
+        "group_norm_weight": true,
+        "hidden_size": 960,
+        "initial_val": 0.0,
+        "input_shape": "BSGNH",
+        "internal_input_shape": "SBNGH",
+        "num_gates": 4,
+        "num_heads": 4,
+        "num_states": 4,
+        "output_shape": "BNSH",
+        "recurrent_weight_init": "zeros"
+      }
+    },
+    "tie_weights": false,
+    "vocab_size": 49152,
+    "weight_decay_on_embedding": false
+  }
+}

events.out.tfevents.1741447692.ef0fed4470ac.1412.0manual_logs ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:669d773e69f4432118654f8c0f2fda720cdc5d7e58b98dea3fb062901ccb3575
+size 5119656

events.out.tfevents.1741447693.ef0fed4470ac.1412.1 ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:f8295c03b48a70a216c0c7bfb4eae4a1bc467227d4126af5b4d16a5187ff6c80
+size 11930

events.out.tfevents.1741450037.ef0fed4470ac.1412.2 ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:6f7aa8fcbeb3920e9f92716963dc064bd9b2e0cd4286687fa4012df14abd4c61
+size 193

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:7ebcffbeb034a8cdc7420fe4ad01b33205cd5d2f3bcec0b1204f3a9fc0c2c67a
+size 761042760

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,34 @@

+{
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>"
+  ],
+  "bos_token": {
+    "content": "<|im_start|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "<|im_end|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<|im_end|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,155 @@

+{
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "<repo_name>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "4": {
+      "content": "<reponame>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "5": {
+      "content": "<file_sep>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "6": {
+      "content": "<filename>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "7": {
+      "content": "<gh_stars>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "8": {
+      "content": "<issue_start>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "9": {
+      "content": "<issue_comment>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "10": {
+      "content": "<issue_closed>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "11": {
+      "content": "<jupyter_start>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "12": {
+      "content": "<jupyter_text>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "13": {
+      "content": "<jupyter_code>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "14": {
+      "content": "<jupyter_output>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "15": {
+      "content": "<jupyter_script>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "16": {
+      "content": "<empty_output>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>"
+  ],
+  "bos_token": "<|im_start|>",
+  "chat_template": "{% for message in messages %}{% if loop.first and messages[0]['role'] != 'system' %}{{ '<|im_start|>system\nYou are a helpful AI assistant named SmolLM, trained by Hugging Face<|im_end|>\n' }}{% endif %}{{'<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>' + '\n'}}{% endfor %}{% if add_generation_prompt %}{{ '<|im_start|>assistant\n' }}{% endif %}",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|im_end|>",
+  "extra_special_tokens": {},
+  "model_max_length": 8192,
+  "pad_token": "<|im_end|>",
+  "tokenizer_class": "GPT2Tokenizer",
+  "unk_token": "<|endoftext|>",
+  "vocab_size": 49152
+}

training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:8c11814ba2b844e5146f38162738d7be033331e33a0d4f4a11158374d806b9d4
+size 6264

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff