iqrakiran commited on
Commit
4dfb6dc
·
verified ·
1 Parent(s): 92fd20d

Upload tokenizer

Browse files
Files changed (2) hide show
  1. special_tokens_map.json +6 -7
  2. tokenizer_config.json +3 -5
special_tokens_map.json CHANGED
@@ -2,26 +2,25 @@
2
  "bos_token": {
3
  "content": "<s>",
4
  "lstrip": false,
5
- "normalized": false,
6
  "rstrip": false,
7
  "single_word": false,
8
- "special": true
9
  },
10
  "eos_token": {
11
  "content": "</s>",
12
  "lstrip": false,
13
- "normalized": false,
14
  "rstrip": false,
15
  "single_word": false,
16
- "special": true
17
  },
18
- "pad_token": "</s>",
19
  "unk_token": {
20
  "content": "<unk>",
21
  "lstrip": false,
22
- "normalized": false,
23
  "rstrip": false,
24
  "single_word": false,
25
- "special": true
26
  }
27
  }
 
2
  "bos_token": {
3
  "content": "<s>",
4
  "lstrip": false,
5
+ "normalized": true,
6
  "rstrip": false,
7
  "single_word": false,
8
+ "special": false
9
  },
10
  "eos_token": {
11
  "content": "</s>",
12
  "lstrip": false,
13
+ "normalized": true,
14
  "rstrip": false,
15
  "single_word": false,
16
+ "special": false
17
  },
 
18
  "unk_token": {
19
  "content": "<unk>",
20
  "lstrip": false,
21
+ "normalized": true,
22
  "rstrip": false,
23
  "single_word": false,
24
+ "special": false
25
  }
26
  }
tokenizer_config.json CHANGED
@@ -5,7 +5,7 @@
5
  "__type": "AddedToken",
6
  "content": "<s>",
7
  "lstrip": false,
8
- "normalized": false,
9
  "rstrip": false,
10
  "single_word": false,
11
  "special": false
@@ -15,22 +15,20 @@
15
  "__type": "AddedToken",
16
  "content": "</s>",
17
  "lstrip": false,
18
- "normalized": false,
19
  "rstrip": false,
20
  "single_word": false,
21
  "special": false
22
  },
23
- "legacy": false,
24
  "model_max_length": 1000000000000000019884624838656,
25
  "pad_token": null,
26
- "padding_side": "right",
27
  "sp_model_kwargs": {},
28
  "tokenizer_class": "LlamaTokenizer",
29
  "unk_token": {
30
  "__type": "AddedToken",
31
  "content": "<unk>",
32
  "lstrip": false,
33
- "normalized": false,
34
  "rstrip": false,
35
  "single_word": false,
36
  "special": false
 
5
  "__type": "AddedToken",
6
  "content": "<s>",
7
  "lstrip": false,
8
+ "normalized": true,
9
  "rstrip": false,
10
  "single_word": false,
11
  "special": false
 
15
  "__type": "AddedToken",
16
  "content": "</s>",
17
  "lstrip": false,
18
+ "normalized": true,
19
  "rstrip": false,
20
  "single_word": false,
21
  "special": false
22
  },
 
23
  "model_max_length": 1000000000000000019884624838656,
24
  "pad_token": null,
 
25
  "sp_model_kwargs": {},
26
  "tokenizer_class": "LlamaTokenizer",
27
  "unk_token": {
28
  "__type": "AddedToken",
29
  "content": "<unk>",
30
  "lstrip": false,
31
+ "normalized": true,
32
  "rstrip": false,
33
  "single_word": false,
34
  "special": false