jan-hq commited on
Commit
328da72
1 Parent(s): 8472057

Upload tokenizer

Browse files
Files changed (4) hide show
  1. README.md +1 -1
  2. special_tokens_map.json +16 -0
  3. tokenizer.json +0 -0
  4. tokenizer_config.json +0 -0
README.md CHANGED
@@ -1,9 +1,9 @@
1
  ---
2
- license: apache-2.0
3
  datasets:
4
  - jan-hq/instruction-speech-v1
5
  language:
6
  - en
 
7
  tags:
8
  - sound language model
9
  ---
 
1
  ---
 
2
  datasets:
3
  - jan-hq/instruction-speech-v1
4
  language:
5
  - en
6
+ license: apache-2.0
7
  tags:
8
  - sound language model
9
  ---
special_tokens_map.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "bos_token": {
3
+ "content": "<|begin_of_text|>",
4
+ "lstrip": false,
5
+ "normalized": false,
6
+ "rstrip": false,
7
+ "single_word": false
8
+ },
9
+ "eos_token": {
10
+ "content": "<|eot_id|>",
11
+ "lstrip": false,
12
+ "normalized": false,
13
+ "rstrip": false,
14
+ "single_word": false
15
+ }
16
+ }
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
The diff for this file is too large to render. See raw diff