Mr-Vicky-01 commited on
Commit
e5ca55f
1 Parent(s): 6cb8e35

Upload tokenizer

Browse files
Files changed (2) hide show
  1. special_tokens_map.json +9 -15
  2. tokenizer_config.json +5 -1
special_tokens_map.json CHANGED
@@ -1,19 +1,7 @@
1
  {
2
  "additional_special_tokens": [
3
- {
4
- "content": "<sos>",
5
- "lstrip": false,
6
- "normalized": false,
7
- "rstrip": false,
8
- "single_word": false
9
- },
10
- {
11
- "content": "<eos>",
12
- "lstrip": false,
13
- "normalized": false,
14
- "rstrip": false,
15
- "single_word": false
16
- }
17
  ],
18
  "bos_token": {
19
  "content": "<|endoftext|>",
@@ -29,7 +17,13 @@
29
  "rstrip": false,
30
  "single_word": false
31
  },
32
- "pad_token": "<|endoftext|>",
 
 
 
 
 
 
33
  "unk_token": {
34
  "content": "<|endoftext|>",
35
  "lstrip": false,
 
1
  {
2
  "additional_special_tokens": [
3
+ "<sos>",
4
+ "<eos>"
 
 
 
 
 
 
 
 
 
 
 
 
5
  ],
6
  "bos_token": {
7
  "content": "<|endoftext|>",
 
17
  "rstrip": false,
18
  "single_word": false
19
  },
20
+ "pad_token": {
21
+ "content": "<|endoftext|>",
22
+ "lstrip": false,
23
+ "normalized": true,
24
+ "rstrip": false,
25
+ "single_word": false
26
+ },
27
  "unk_token": {
28
  "content": "<|endoftext|>",
29
  "lstrip": false,
tokenizer_config.json CHANGED
@@ -41,12 +41,16 @@
41
  "<eos>"
42
  ],
43
  "bos_token": "<|endoftext|>",
44
- "clean_up_tokenization_spaces": false,
45
  "eos_token": "<|endoftext|>",
46
  "errors": "replace",
 
47
  "model_max_length": 512,
48
  "pad_token": "<|endoftext|>",
49
  "padding_side": "right",
 
50
  "tokenizer_class": "GPT2Tokenizer",
 
 
51
  "unk_token": "<|endoftext|>"
52
  }
 
41
  "<eos>"
42
  ],
43
  "bos_token": "<|endoftext|>",
44
+ "clean_up_tokenization_spaces": true,
45
  "eos_token": "<|endoftext|>",
46
  "errors": "replace",
47
+ "max_length": 512,
48
  "model_max_length": 512,
49
  "pad_token": "<|endoftext|>",
50
  "padding_side": "right",
51
+ "stride": 0,
52
  "tokenizer_class": "GPT2Tokenizer",
53
+ "truncation_side": "right",
54
+ "truncation_strategy": "longest_first",
55
  "unk_token": "<|endoftext|>"
56
  }