bishaltwr commited on
Commit
a80e90d
·
verified ·
1 Parent(s): 3427c2b

Upload tokenizer

Browse files
added_tokens.json CHANGED
@@ -1,4 +1,4 @@
1
  {
2
- "</s>": 71,
3
- "<s>": 70
4
  }
 
1
  {
2
+ "</s>": 94,
3
+ "<s>": 93
4
  }
special_tokens_map.json CHANGED
@@ -1,30 +1,6 @@
1
  {
2
- "bos_token": {
3
- "content": "<s>",
4
- "lstrip": false,
5
- "normalized": false,
6
- "rstrip": false,
7
- "single_word": false
8
- },
9
- "eos_token": {
10
- "content": "</s>",
11
- "lstrip": false,
12
- "normalized": false,
13
- "rstrip": false,
14
- "single_word": false
15
- },
16
- "pad_token": {
17
- "content": "[PAD]",
18
- "lstrip": true,
19
- "normalized": false,
20
- "rstrip": true,
21
- "single_word": false
22
- },
23
- "unk_token": {
24
- "content": "[UNK]",
25
- "lstrip": true,
26
- "normalized": false,
27
- "rstrip": true,
28
- "single_word": false
29
- }
30
  }
 
1
  {
2
+ "bos_token": "<s>",
3
+ "eos_token": "</s>",
4
+ "pad_token": "[PAD]",
5
+ "unk_token": "[UNK]"
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
6
  }
tokenizer_config.json CHANGED
@@ -1,6 +1,6 @@
1
  {
2
  "added_tokens_decoder": {
3
- "68": {
4
  "content": "[UNK]",
5
  "lstrip": true,
6
  "normalized": false,
@@ -8,7 +8,7 @@
8
  "single_word": false,
9
  "special": false
10
  },
11
- "69": {
12
  "content": "[PAD]",
13
  "lstrip": true,
14
  "normalized": false,
@@ -16,7 +16,7 @@
16
  "single_word": false,
17
  "special": false
18
  },
19
- "70": {
20
  "content": "<s>",
21
  "lstrip": false,
22
  "normalized": false,
@@ -24,7 +24,7 @@
24
  "single_word": false,
25
  "special": true
26
  },
27
- "71": {
28
  "content": "</s>",
29
  "lstrip": false,
30
  "normalized": false,
 
1
  {
2
  "added_tokens_decoder": {
3
+ "91": {
4
  "content": "[UNK]",
5
  "lstrip": true,
6
  "normalized": false,
 
8
  "single_word": false,
9
  "special": false
10
  },
11
+ "92": {
12
  "content": "[PAD]",
13
  "lstrip": true,
14
  "normalized": false,
 
16
  "single_word": false,
17
  "special": false
18
  },
19
+ "93": {
20
  "content": "<s>",
21
  "lstrip": false,
22
  "normalized": false,
 
24
  "single_word": false,
25
  "special": true
26
  },
27
+ "94": {
28
  "content": "</s>",
29
  "lstrip": false,
30
  "normalized": false,
vocab.json CHANGED
@@ -3,72 +3,95 @@
3
  "(": 1,
4
  ")": 2,
5
  "/": 3,
6
- "[PAD]": 69,
7
- "[UNK]": 68,
 
 
 
 
 
 
 
8
  "|": 0,
9
- "ँ": 4,
10
- "ं": 5,
11
- "ः": 6,
12
- "अ": 7,
13
- "आ": 8,
14
- "इ": 9,
15
- "ई": 10,
16
- "उ": 11,
17
- "ऊ": 12,
18
- "ऋ": 13,
19
- "ए": 14,
20
- "ऐ": 15,
21
- "ओ": 16,
22
- "औ": 17,
23
- "क": 18,
24
- "ख": 19,
25
- "ग": 20,
26
- "घ": 21,
27
- "ङ": 22,
28
- "च": 23,
29
- "छ": 24,
30
- "ज": 25,
31
- "झ": 26,
32
- "ञ": 27,
33
- "ट": 28,
34
- "ठ": 29,
35
- "ड": 30,
36
- "ढ": 31,
37
- "ण": 32,
38
- "त": 33,
39
- "थ": 34,
40
- "द": 35,
41
- "ध": 36,
42
- "न": 37,
43
- "प": 38,
44
- "फ": 39,
45
- "ब": 40,
46
- "भ": 41,
47
- "म": 42,
48
- "य": 43,
49
- "र": 44,
50
- "ऱ": 45,
51
- "ल": 46,
52
- "व": 47,
53
- "श": 48,
54
- "ष": 49,
55
- "स": 50,
56
- "ह": 51,
57
- "": 52,
58
- "ि": 53,
59
- "": 54,
60
- "": 55,
61
- "": 56,
62
- "": 57,
63
- "": 58,
64
- "": 59,
65
- "": 60,
66
- "": 61,
67
- "": 62,
68
- "": 63,
69
- "": 64,
70
- "": 65,
71
- "": 66,
72
- "": 67
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
73
  }
74
  }
 
3
  "(": 1,
4
  ")": 2,
5
  "/": 3,
6
+ "[PAD]": 92,
7
+ "[UNK]": 91,
8
+ "a": 4,
9
+ "b": 5,
10
+ "c": 6,
11
+ "e": 7,
12
+ "f": 8,
13
+ "k": 9,
14
+ "o": 10,
15
  "|": 0,
16
+ "ँ": 11,
17
+ "ं": 12,
18
+ "ः": 13,
19
+ "अ": 14,
20
+ "आ": 15,
21
+ "इ": 16,
22
+ "ई": 17,
23
+ "उ": 18,
24
+ "ऊ": 19,
25
+ "ऋ": 20,
26
+ "ए": 21,
27
+ "ऐ": 22,
28
+ "ओ": 23,
29
+ "औ": 24,
30
+ "क": 25,
31
+ "ख": 26,
32
+ "ग": 27,
33
+ "घ": 28,
34
+ "ङ": 29,
35
+ "च": 30,
36
+ "छ": 31,
37
+ "ज": 32,
38
+ "झ": 33,
39
+ "ञ": 34,
40
+ "ट": 35,
41
+ "ठ": 36,
42
+ "ड": 37,
43
+ "ढ": 38,
44
+ "ण": 39,
45
+ "त": 40,
46
+ "थ": 41,
47
+ "द": 42,
48
+ "ध": 43,
49
+ "न": 44,
50
+ "प": 45,
51
+ "फ": 46,
52
+ "ब": 47,
53
+ "भ": 48,
54
+ "म": 49,
55
+ "य": 50,
56
+ "र": 51,
57
+ "ऱ": 52,
58
+ "ल": 53,
59
+ "व": 54,
60
+ "श": 55,
61
+ "ष": 56,
62
+ "स": 57,
63
+ "ह": 58,
64
+ "": 59,
65
+ "": 60,
66
+ "ि": 61,
67
+ "": 62,
68
+ "": 63,
69
+ "": 64,
70
+ "": 65,
71
+ "": 66,
72
+ "": 67,
73
+ "": 68,
74
+ "": 69,
75
+ "": 70,
76
+ "": 71,
77
+ "": 72,
78
+ "": 73,
79
+ "": 74,
80
+ "१": 75,
81
+ "२": 76,
82
+ "३": 77,
83
+ "४": 78,
84
+ "५": 79,
85
+ "६": 80,
86
+ "७": 81,
87
+ "८": 82,
88
+ "९": 83,
89
+ "॰": 84,
90
+ "‌": 85,
91
+ "‍": 86,
92
+ "‎": 87,
93
+ "–": 88,
94
+ "—": 89,
95
+ "’": 90
96
  }
97
  }