rajammanabrolu commited on
Commit
772b643
1 Parent(s): b519b19

Upload tokenizer

Browse files
Files changed (2) hide show
  1. special_tokens_map.json +3 -21
  2. tokenizer_config.json +0 -66
special_tokens_map.json CHANGED
@@ -3,26 +3,8 @@
3
  "<|im_start|>",
4
  "<|im_end|>"
5
  ],
6
- "bos_token": {
7
- "content": "<|endoftext|>",
8
- "lstrip": false,
9
- "normalized": false,
10
- "rstrip": false,
11
- "single_word": false
12
- },
13
- "eos_token": {
14
- "content": "<|endoftext|>",
15
- "lstrip": false,
16
- "normalized": false,
17
- "rstrip": false,
18
- "single_word": false
19
- },
20
  "pad_token": "<|pad|>",
21
- "unk_token": {
22
- "content": "<|endoftext|>",
23
- "lstrip": false,
24
- "normalized": false,
25
- "rstrip": false,
26
- "single_word": false
27
- }
28
  }
 
3
  "<|im_start|>",
4
  "<|im_end|>"
5
  ],
6
+ "bos_token": "<|endoftext|>",
7
+ "eos_token": "<|endoftext|>",
 
 
 
 
 
 
 
 
 
 
 
 
8
  "pad_token": "<|pad|>",
9
+ "unk_token": "<|endoftext|>"
 
 
 
 
 
 
10
  }
tokenizer_config.json CHANGED
@@ -2,72 +2,6 @@
2
  "add_bos_token": false,
3
  "add_eos_token": false,
4
  "add_prefix_space": false,
5
- "added_tokens_decoder": {
6
- "100257": {
7
- "content": "<|endoftext|>",
8
- "lstrip": false,
9
- "normalized": false,
10
- "rstrip": false,
11
- "single_word": false,
12
- "special": true
13
- },
14
- "100258": {
15
- "content": "<|fim_prefix|>",
16
- "lstrip": false,
17
- "normalized": false,
18
- "rstrip": false,
19
- "single_word": false,
20
- "special": true
21
- },
22
- "100259": {
23
- "content": "<|fim_middle|>",
24
- "lstrip": false,
25
- "normalized": false,
26
- "rstrip": false,
27
- "single_word": false,
28
- "special": true
29
- },
30
- "100260": {
31
- "content": "<|fim_suffix|>",
32
- "lstrip": false,
33
- "normalized": false,
34
- "rstrip": false,
35
- "single_word": false,
36
- "special": true
37
- },
38
- "100276": {
39
- "content": "<|endofprompt|>",
40
- "lstrip": false,
41
- "normalized": false,
42
- "rstrip": false,
43
- "single_word": false,
44
- "special": true
45
- },
46
- "100277": {
47
- "content": "<|pad|>",
48
- "lstrip": false,
49
- "normalized": false,
50
- "rstrip": false,
51
- "single_word": false,
52
- "special": true
53
- },
54
- "100278": {
55
- "content": "<|im_start|>",
56
- "lstrip": false,
57
- "normalized": false,
58
- "rstrip": false,
59
- "single_word": false,
60
- "special": true
61
- },
62
- "100279": {
63
- "content": "<|im_end|>",
64
- "lstrip": false,
65
- "normalized": false,
66
- "rstrip": false,
67
- "single_word": false,
68
- "special": true
69
- }
70
- },
71
  "additional_special_tokens": [
72
  "<|im_start|>",
73
  "<|im_end|>"
 
2
  "add_bos_token": false,
3
  "add_eos_token": false,
4
  "add_prefix_space": false,
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5
  "additional_special_tokens": [
6
  "<|im_start|>",
7
  "<|im_end|>"