qubitpage commited on
Commit
43334d8
·
verified ·
1 Parent(s): dd4f389

Upload tokenizer

Browse files
Files changed (2) hide show
  1. tiktoken_vocab.json +8 -0
  2. tokenizer_config.json +17 -0
tiktoken_vocab.json ADDED
@@ -0,0 +1,8 @@
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "encoding": "cl100k_base",
3
+ "vocab_size": 100277,
4
+ "eos_token_id": 100257,
5
+ "special_tokens": {
6
+ "<|endoftext|>": 100257
7
+ }
8
+ }
tokenizer_config.json ADDED
@@ -0,0 +1,17 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "added_tokens_decoder": {
3
+ "100257": {
4
+ "content": "<|endoftext|>",
5
+ "lstrip": false,
6
+ "normalized": false,
7
+ "rstrip": false,
8
+ "single_word": false,
9
+ "special": true
10
+ }
11
+ },
12
+ "backend": "custom",
13
+ "eos_token": "<|endoftext|>",
14
+ "model_max_length": 2048,
15
+ "pad_token": "<|endoftext|>",
16
+ "tokenizer_class": "SentinelBrainTokenizer"
17
+ }