i4never commited on
Commit
28ff41f
1 Parent(s): e8f2a21

Upload tokenizer

Browse files
Files changed (3) hide show
  1. .gitattributes +1 -0
  2. tokenizer.json +3 -0
  3. tokenizer_config.json +1 -1
.gitattributes CHANGED
@@ -32,3 +32,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
32
  *.zip filter=lfs diff=lfs merge=lfs -text
33
  *.zst filter=lfs diff=lfs merge=lfs -text
34
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
32
  *.zip filter=lfs diff=lfs merge=lfs -text
33
  *.zst filter=lfs diff=lfs merge=lfs -text
34
  *tfevents* filter=lfs diff=lfs merge=lfs -text
35
+ tokenizer.json filter=lfs diff=lfs merge=lfs -text
tokenizer.json ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:52b081922226132b1e8408182a64d69779463959ecad8097a0333c3923e7cd74
3
+ size 14500838
tokenizer_config.json CHANGED
@@ -1,11 +1,11 @@
1
  {
2
  "add_prefix_space": false,
3
  "bos_token": "<s>",
 
4
  "eos_token": "</s>",
5
  "model_max_length": 1024,
6
  "pad_token": "<pad>",
7
  "padding_side": "right",
8
- "special_tokens_map_file": null,
9
  "tokenizer_class": "BloomTokenizer",
10
  "trunction_side": "right",
11
  "unk_token": "<unk>"
 
1
  {
2
  "add_prefix_space": false,
3
  "bos_token": "<s>",
4
+ "clean_up_tokenization_spaces": false,
5
  "eos_token": "</s>",
6
  "model_max_length": 1024,
7
  "pad_token": "<pad>",
8
  "padding_side": "right",
 
9
  "tokenizer_class": "BloomTokenizer",
10
  "trunction_side": "right",
11
  "unk_token": "<unk>"