Training in progress, step 500

Browse files

Files changed (10) hide show

config.json +96 -0
events.out.tfevents.1741458086.dfca901dbc5d.9483.0 +3 -0
merges.txt +0 -0
model.safetensors +3 -0
runs/Mar08_18-02-19_dfca901dbc5d/events.out.tfevents.1741456943.dfca901dbc5d.4335.0 +3 -0
special_tokens_map.json +34 -0
tokenizer.json +0 -0
tokenizer_config.json +155 -0
training_args.bin +3 -0
vocab.json +0 -0

config.json ADDED Viewed

	@@ -0,0 +1,96 @@

+{
+  "architectures": [
+    "DistilxLSTM"
+  ],
+  "model_type": "xlstm",
+  "pad_token_id": 2,
+  "torch_dtype": "float32",
+  "transformers_version": "4.48.3",
+  "xlstm_cfg": {
+    "_block_map": "1,0,0,0,1,0,0,0,1,0,0,0,1,0,0,0",
+    "add_embedding_dropout": false,
+    "add_post_blocks_norm": true,
+    "bias": false,
+    "context_length": 256,
+    "dropout": 0.0,
+    "embedding_dim": 960,
+    "mlstm_block": {
+      "_block_idx": null,
+      "_num_blocks": 16,
+      "mlstm": {
+        "_inner_embedding_dim": 1920,
+        "_num_blocks": 16,
+        "_proj_up_dim": 1920,
+        "bias": false,
+        "context_length": 256,
+        "conv1d_kernel_size": 4,
+        "dropout": 0.0,
+        "embedding_dim": 960,
+        "num_heads": 4,
+        "proj_factor": 2.0,
+        "qkv_proj_blocksize": 32,
+        "round_proj_up_dim_up": true,
+        "round_proj_up_to_multiple_of": 64
+      }
+    },
+    "num_blocks": 16,
+    "slstm_at": [
+      0,
+      4,
+      8,
+      12
+    ],
+    "slstm_block": {
+      "_block_idx": null,
+      "_num_blocks": 16,
+      "feedforward": {
+        "_num_blocks": 1,
+        "_proj_up_dim": 0,
+        "act_fn": "gelu",
+        "bias": false,
+        "dropout": 0.0,
+        "embedding_dim": -1,
+        "ff_type": "ffn_gated",
+        "proj_factor": 1.7,
+        "round_proj_up_dim_up": true,
+        "round_proj_up_to_multiple_of": 64
+      },
+      "slstm": {
+        "_block_idx": null,
+        "_num_blocks": 16,
+        "backend": "cuda",
+        "batch_size": 8,
+        "bias_init": "powerlaw_blockdependent",
+        "constants": {},
+        "conv1d_kernel_size": 4,
+        "dropout": 0.0,
+        "dtype": "bfloat16",
+        "dtype_a": "float32",
+        "dtype_b": "float32",
+        "dtype_g": "bfloat16",
+        "dtype_r": "bfloat16",
+        "dtype_s": "bfloat16",
+        "dtype_w": "bfloat16",
+        "embedding_dim": 960,
+        "enable_automatic_mixed_precision": true,
+        "forward_clipval": null,
+        "function": "slstm",
+        "gradient_recurrent_clipval": null,
+        "gradient_recurrent_cut": false,
+        "group_norm_weight": true,
+        "hidden_size": 960,
+        "initial_val": 0.0,
+        "input_shape": "BSGNH",
+        "internal_input_shape": "SBNGH",
+        "num_gates": 4,
+        "num_heads": 4,
+        "num_states": 4,
+        "output_shape": "BNSH",
+        "recurrent_weight_init": "zeros"
+      }
+    },
+    "tie_weights": false,
+    "vocab_size": 49152,
+    "weight_decay_on_embedding": false
+  }
+}

events.out.tfevents.1741458086.dfca901dbc5d.9483.0 ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:c02f204248f3bdae8f4b3ad92d753b7f692277ef7e8bf4c083d5b102a13e9bae
+size 7321

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:9c75fce354a35d9674b5b7f7eeab3fec2e2aad29d3e51cf53ec74970853970ac
+size 761042760

runs/Mar08_18-02-19_dfca901dbc5d/events.out.tfevents.1741456943.dfca901dbc5d.4335.0 ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:51cc0a4ffe5b3f7f25e0280aa94765dbf906645bcf410d3ab68e3645893d5a62
+size 6885

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,34 @@

+{
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>"
+  ],
+  "bos_token": {
+    "content": "<|im_start|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "<|im_end|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<|im_end|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,155 @@

+{
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<|im_start|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "<|im_end|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "<repo_name>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "4": {
+      "content": "<reponame>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "5": {
+      "content": "<file_sep>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "6": {
+      "content": "<filename>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "7": {
+      "content": "<gh_stars>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "8": {
+      "content": "<issue_start>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "9": {
+      "content": "<issue_comment>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "10": {
+      "content": "<issue_closed>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "11": {
+      "content": "<jupyter_start>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "12": {
+      "content": "<jupyter_text>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "13": {
+      "content": "<jupyter_code>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "14": {
+      "content": "<jupyter_output>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "15": {
+      "content": "<jupyter_script>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "16": {
+      "content": "<empty_output>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "additional_special_tokens": [
+    "<|im_start|>",
+    "<|im_end|>"
+  ],
+  "bos_token": "<|im_start|>",
+  "chat_template": "{% for message in messages %}{% if loop.first and messages[0]['role'] != 'system' %}{{ '<|im_start|>system\nYou are a helpful AI assistant named SmolLM, trained by Hugging Face<|im_end|>\n' }}{% endif %}{{'<|im_start|>' + message['role'] + '\n' + message['content'] + '<|im_end|>' + '\n'}}{% endfor %}{% if add_generation_prompt %}{{ '<|im_start|>assistant\n' }}{% endif %}",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|im_end|>",
+  "extra_special_tokens": {},
+  "model_max_length": 8192,
+  "pad_token": "<|im_end|>",
+  "tokenizer_class": "GPT2Tokenizer",
+  "unk_token": "<|endoftext|>",
+  "vocab_size": 49152
+}

training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:85b7c97cb04bd72bab46999b30ca5d2928241d8d143898bd70eee14e5e2bfa91
+size 5368

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff