anthonyrathe commited on
Commit
0bafe8e
·
verified ·
1 Parent(s): 55290a4

Update tokenizer

Browse files
Files changed (1) hide show
  1. tokenizer_config.json +1 -4
tokenizer_config.json CHANGED
@@ -45,13 +45,10 @@
45
  "bos_token": "<s>",
46
  "chat_template": "{% for message in messages %}\n{% if message['role'] == 'user' %}\n{{ '<|user|>\n' + message['content'] + eos_token }}\n{% elif message['role'] == 'system' %}\n{{ '<|system|>\n' + message['content'] + eos_token }}\n{% elif message['role'] == 'assistant' %}\n{{ '<|assistant|>\n' + message['content'] + eos_token }}\n{% endif %}\n{% if loop.last and add_generation_prompt %}\n{{ '<|assistant|>' }}\n{% endif %}\n{% endfor %}",
47
  "clean_up_tokenization_spaces": true,
48
- "cls_token": "<s>",
49
  "eos_token": "</s>",
50
  "errors": "replace",
51
- "mask_token": "<mask>",
52
- "model_max_length": 512,
53
  "pad_token": "<pad>",
54
- "sep_token": "</s>",
55
  "tokenizer_class": "RobertaTokenizer",
56
  "trim_offsets": true,
57
  "unk_token": "<unk>"
 
45
  "bos_token": "<s>",
46
  "chat_template": "{% for message in messages %}\n{% if message['role'] == 'user' %}\n{{ '<|user|>\n' + message['content'] + eos_token }}\n{% elif message['role'] == 'system' %}\n{{ '<|system|>\n' + message['content'] + eos_token }}\n{% elif message['role'] == 'assistant' %}\n{{ '<|assistant|>\n' + message['content'] + eos_token }}\n{% endif %}\n{% if loop.last and add_generation_prompt %}\n{{ '<|assistant|>' }}\n{% endif %}\n{% endfor %}",
47
  "clean_up_tokenization_spaces": true,
 
48
  "eos_token": "</s>",
49
  "errors": "replace",
50
+ "model_max_length": 2048,
 
51
  "pad_token": "<pad>",
 
52
  "tokenizer_class": "RobertaTokenizer",
53
  "trim_offsets": true,
54
  "unk_token": "<unk>"