caiom commited on
Commit
2fcbc9b
·
1 Parent(s): a9bfb32

Upload tokenizer

Browse files
added_tokens.json CHANGED
@@ -1,3 +1,14 @@
1
  {
2
- "<|endoftext|>": 32000
 
 
 
 
 
 
 
 
 
 
 
3
  }
 
1
  {
2
+ "<|assistant|>": 32001,
3
+ "<|continue|>": 32009,
4
+ "<|endoftext|>": 32000,
5
+ "<|end|>": 32007,
6
+ "<|function_call|>": 32005,
7
+ "<|function_list|>": 32011,
8
+ "<|function_output|>": 32003,
9
+ "<|raw|>": 32008,
10
+ "<|step|>": 32002,
11
+ "<|system|>": 32006,
12
+ "<|tag|>": 32004,
13
+ "<|user|>": 32010
14
  }
special_tokens_map.json CHANGED
@@ -1,4 +1,83 @@
1
  {
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
2
  "bos_token": {
3
  "content": "<s>",
4
  "lstrip": false,
 
1
  {
2
+ "additional_special_tokens": [
3
+ {
4
+ "content": "<|assistant|>",
5
+ "lstrip": false,
6
+ "normalized": false,
7
+ "rstrip": true,
8
+ "single_word": false
9
+ },
10
+ {
11
+ "content": "<|step|>",
12
+ "lstrip": false,
13
+ "normalized": false,
14
+ "rstrip": true,
15
+ "single_word": false
16
+ },
17
+ {
18
+ "content": "<|function_output|>",
19
+ "lstrip": false,
20
+ "normalized": false,
21
+ "rstrip": true,
22
+ "single_word": false
23
+ },
24
+ {
25
+ "content": "<|tag|>",
26
+ "lstrip": false,
27
+ "normalized": false,
28
+ "rstrip": true,
29
+ "single_word": false
30
+ },
31
+ {
32
+ "content": "<|function_call|>",
33
+ "lstrip": false,
34
+ "normalized": false,
35
+ "rstrip": true,
36
+ "single_word": false
37
+ },
38
+ {
39
+ "content": "<|system|>",
40
+ "lstrip": false,
41
+ "normalized": false,
42
+ "rstrip": true,
43
+ "single_word": false
44
+ },
45
+ {
46
+ "content": "<|end|>",
47
+ "lstrip": false,
48
+ "normalized": false,
49
+ "rstrip": true,
50
+ "single_word": false
51
+ },
52
+ {
53
+ "content": "<|raw|>",
54
+ "lstrip": false,
55
+ "normalized": false,
56
+ "rstrip": true,
57
+ "single_word": false
58
+ },
59
+ {
60
+ "content": "<|continue|>",
61
+ "lstrip": false,
62
+ "normalized": false,
63
+ "rstrip": true,
64
+ "single_word": false
65
+ },
66
+ {
67
+ "content": "<|user|>",
68
+ "lstrip": false,
69
+ "normalized": false,
70
+ "rstrip": true,
71
+ "single_word": false
72
+ },
73
+ {
74
+ "content": "<|function_list|>",
75
+ "lstrip": false,
76
+ "normalized": false,
77
+ "rstrip": true,
78
+ "single_word": false
79
+ }
80
+ ],
81
  "bos_token": {
82
  "content": "<s>",
83
  "lstrip": false,
tokenizer.json CHANGED
@@ -38,6 +38,105 @@
38
  "rstrip": false,
39
  "normalized": false,
40
  "special": true
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
41
  }
42
  ],
43
  "normalizer": {
 
38
  "rstrip": false,
39
  "normalized": false,
40
  "special": true
41
+ },
42
+ {
43
+ "id": 32001,
44
+ "content": "<|assistant|>",
45
+ "single_word": false,
46
+ "lstrip": false,
47
+ "rstrip": true,
48
+ "normalized": false,
49
+ "special": true
50
+ },
51
+ {
52
+ "id": 32002,
53
+ "content": "<|step|>",
54
+ "single_word": false,
55
+ "lstrip": false,
56
+ "rstrip": true,
57
+ "normalized": false,
58
+ "special": true
59
+ },
60
+ {
61
+ "id": 32003,
62
+ "content": "<|function_output|>",
63
+ "single_word": false,
64
+ "lstrip": false,
65
+ "rstrip": true,
66
+ "normalized": false,
67
+ "special": true
68
+ },
69
+ {
70
+ "id": 32004,
71
+ "content": "<|tag|>",
72
+ "single_word": false,
73
+ "lstrip": false,
74
+ "rstrip": true,
75
+ "normalized": false,
76
+ "special": true
77
+ },
78
+ {
79
+ "id": 32005,
80
+ "content": "<|function_call|>",
81
+ "single_word": false,
82
+ "lstrip": false,
83
+ "rstrip": true,
84
+ "normalized": false,
85
+ "special": true
86
+ },
87
+ {
88
+ "id": 32006,
89
+ "content": "<|system|>",
90
+ "single_word": false,
91
+ "lstrip": false,
92
+ "rstrip": true,
93
+ "normalized": false,
94
+ "special": true
95
+ },
96
+ {
97
+ "id": 32007,
98
+ "content": "<|end|>",
99
+ "single_word": false,
100
+ "lstrip": false,
101
+ "rstrip": true,
102
+ "normalized": false,
103
+ "special": true
104
+ },
105
+ {
106
+ "id": 32008,
107
+ "content": "<|raw|>",
108
+ "single_word": false,
109
+ "lstrip": false,
110
+ "rstrip": true,
111
+ "normalized": false,
112
+ "special": true
113
+ },
114
+ {
115
+ "id": 32009,
116
+ "content": "<|continue|>",
117
+ "single_word": false,
118
+ "lstrip": false,
119
+ "rstrip": true,
120
+ "normalized": false,
121
+ "special": true
122
+ },
123
+ {
124
+ "id": 32010,
125
+ "content": "<|user|>",
126
+ "single_word": false,
127
+ "lstrip": false,
128
+ "rstrip": true,
129
+ "normalized": false,
130
+ "special": true
131
+ },
132
+ {
133
+ "id": 32011,
134
+ "content": "<|function_list|>",
135
+ "single_word": false,
136
+ "lstrip": false,
137
+ "rstrip": true,
138
+ "normalized": false,
139
+ "special": true
140
  }
141
  ],
142
  "normalizer": {
tokenizer_config.json CHANGED
@@ -31,8 +31,109 @@
31
  "rstrip": false,
32
  "single_word": false,
33
  "special": true
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
34
  }
35
  },
 
 
 
 
 
 
 
 
 
 
 
 
 
36
  "bos_token": "<s>",
37
  "clean_up_tokenization_spaces": false,
38
  "eos_token": "<|endoftext|>",
 
31
  "rstrip": false,
32
  "single_word": false,
33
  "special": true
34
+ },
35
+ "32001": {
36
+ "content": "<|assistant|>",
37
+ "lstrip": false,
38
+ "normalized": false,
39
+ "rstrip": true,
40
+ "single_word": false,
41
+ "special": true
42
+ },
43
+ "32002": {
44
+ "content": "<|step|>",
45
+ "lstrip": false,
46
+ "normalized": false,
47
+ "rstrip": true,
48
+ "single_word": false,
49
+ "special": true
50
+ },
51
+ "32003": {
52
+ "content": "<|function_output|>",
53
+ "lstrip": false,
54
+ "normalized": false,
55
+ "rstrip": true,
56
+ "single_word": false,
57
+ "special": true
58
+ },
59
+ "32004": {
60
+ "content": "<|tag|>",
61
+ "lstrip": false,
62
+ "normalized": false,
63
+ "rstrip": true,
64
+ "single_word": false,
65
+ "special": true
66
+ },
67
+ "32005": {
68
+ "content": "<|function_call|>",
69
+ "lstrip": false,
70
+ "normalized": false,
71
+ "rstrip": true,
72
+ "single_word": false,
73
+ "special": true
74
+ },
75
+ "32006": {
76
+ "content": "<|system|>",
77
+ "lstrip": false,
78
+ "normalized": false,
79
+ "rstrip": true,
80
+ "single_word": false,
81
+ "special": true
82
+ },
83
+ "32007": {
84
+ "content": "<|end|>",
85
+ "lstrip": false,
86
+ "normalized": false,
87
+ "rstrip": true,
88
+ "single_word": false,
89
+ "special": true
90
+ },
91
+ "32008": {
92
+ "content": "<|raw|>",
93
+ "lstrip": false,
94
+ "normalized": false,
95
+ "rstrip": true,
96
+ "single_word": false,
97
+ "special": true
98
+ },
99
+ "32009": {
100
+ "content": "<|continue|>",
101
+ "lstrip": false,
102
+ "normalized": false,
103
+ "rstrip": true,
104
+ "single_word": false,
105
+ "special": true
106
+ },
107
+ "32010": {
108
+ "content": "<|user|>",
109
+ "lstrip": false,
110
+ "normalized": false,
111
+ "rstrip": true,
112
+ "single_word": false,
113
+ "special": true
114
+ },
115
+ "32011": {
116
+ "content": "<|function_list|>",
117
+ "lstrip": false,
118
+ "normalized": false,
119
+ "rstrip": true,
120
+ "single_word": false,
121
+ "special": true
122
  }
123
  },
124
+ "additional_special_tokens": [
125
+ "<|assistant|>",
126
+ "<|step|>",
127
+ "<|function_output|>",
128
+ "<|tag|>",
129
+ "<|function_call|>",
130
+ "<|system|>",
131
+ "<|end|>",
132
+ "<|raw|>",
133
+ "<|continue|>",
134
+ "<|user|>",
135
+ "<|function_list|>"
136
+ ],
137
  "bos_token": "<s>",
138
  "clean_up_tokenization_spaces": false,
139
  "eos_token": "<|endoftext|>",