michaelbenayoun HF Staff commited on
Commit
157d051
·
verified ·
1 Parent(s): 2856e0f

Upload tokenizer

Browse files
special_tokens_map.json CHANGED
@@ -1,37 +1,20 @@
1
  {
2
- "additional_special_tokens": [
3
- "<|im_end|>",
4
- "<|im_user|>",
5
- "<|im_assistant|>",
6
- "<|start_header_id|>",
7
- "<|end_header_id|>",
8
- "[EOT]",
9
- "<|im_system|>",
10
- "<|im_middle|>"
11
- ],
12
  "bos_token": {
13
- "content": "[BOS]",
14
  "lstrip": false,
15
  "normalized": false,
16
  "rstrip": false,
17
  "single_word": false
18
  },
19
  "eos_token": {
20
- "content": "[EOS]",
21
  "lstrip": false,
22
  "normalized": false,
23
  "rstrip": false,
24
  "single_word": false
25
  },
26
  "pad_token": {
27
- "content": "[PAD]",
28
- "lstrip": false,
29
- "normalized": false,
30
- "rstrip": false,
31
- "single_word": false
32
- },
33
- "unk_token": {
34
- "content": "[UNK]",
35
  "lstrip": false,
36
  "normalized": false,
37
  "rstrip": false,
 
1
  {
 
 
 
 
 
 
 
 
 
 
2
  "bos_token": {
3
+ "content": "<|begin▁of▁sentence|>",
4
  "lstrip": false,
5
  "normalized": false,
6
  "rstrip": false,
7
  "single_word": false
8
  },
9
  "eos_token": {
10
+ "content": "<|end▁of▁sentence|>",
11
  "lstrip": false,
12
  "normalized": false,
13
  "rstrip": false,
14
  "single_word": false
15
  },
16
  "pad_token": {
17
+ "content": "<|end▁of▁sentence|>",
 
 
 
 
 
 
 
18
  "lstrip": false,
19
  "normalized": false,
20
  "rstrip": false,
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json CHANGED
The diff for this file is too large to render. See raw diff