zhangir-azerbayev commited on
Commit
87c9edc
β€’
1 Parent(s): 5a4995c

fix tokenizer

Browse files
Files changed (3) hide show
  1. tokenizer.json +2 -29
  2. tokenizer.model +2 -2
  3. tokenizer_config.json +5 -4
tokenizer.json CHANGED
@@ -32134,23 +32134,7 @@
32134
  "μ™•": 31996,
32135
  "ζ”Ά": 31997,
32136
  "弘": 31998,
32137
- "η»™": 31999,
32138
- "▁<SU": 32000,
32139
- "▁<SUF": 32001,
32140
- "▁<PRE": 32002,
32141
- "▁<M": 32003,
32142
- "▁<MID": 32004,
32143
- "▁<E": 32005,
32144
- "▁<EOT": 32006,
32145
- "▁<PRE>": 32007,
32146
- "▁<SUF>": 32008,
32147
- "▁<MID>": 32009,
32148
- "▁<EOT>": 32010,
32149
- "▁<EOT><EOT>": 32011,
32150
- "▁<EOT><EOT><EOT>": 32012,
32151
- "▁<EOT><EOT><EOT><EOT>": 32013,
32152
- "▁<EOT><EOT><EOT><EOT><EOT>": 32014,
32153
- "▁<EOT><EOT><EOT><EOT><EOT><EOT>": 32015
32154
  },
32155
  "merges": [
32156
  "▁ t",
@@ -93401,18 +93385,7 @@
93401
  "▁▁▁▁▁▁▁▁▁ ▁▁▁▁▁▁",
93402
  "▁▁▁▁▁▁▁ ▁▁▁▁▁▁▁▁",
93403
  "▁▁▁▁▁▁▁▁▁▁▁ ▁▁▁▁",
93404
- "▁ ▁▁▁▁▁▁▁▁▁▁▁▁▁▁",
93405
- "▁< SU",
93406
- "▁<SU F",
93407
- "▁< PRE",
93408
- "▁< M",
93409
- "▁<M ID",
93410
- "▁< E",
93411
- "▁<E OT",
93412
- "▁<PRE >",
93413
- "▁<SUF >",
93414
- "▁<MID >",
93415
- "▁<EOT >"
93416
  ]
93417
  }
93418
  }
 
32134
  "μ™•": 31996,
32135
  "ζ”Ά": 31997,
32136
  "弘": 31998,
32137
+ "η»™": 31999
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
32138
  },
32139
  "merges": [
32140
  "▁ t",
 
93385
  "▁▁▁▁▁▁▁▁▁ ▁▁▁▁▁▁",
93386
  "▁▁▁▁▁▁▁ ▁▁▁▁▁▁▁▁",
93387
  "▁▁▁▁▁▁▁▁▁▁▁ ▁▁▁▁",
93388
+ "▁ ▁▁▁▁▁▁▁▁▁▁▁▁▁▁"
 
 
 
 
 
 
 
 
 
 
 
93389
  ]
93390
  }
93391
  }
tokenizer.model CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:45ccb9c8b6b561889acea59191d66986d314e7cbd6a78abc6e49b139ca91c1e6
3
- size 500058
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9e556afd44213b6bd1be2b850ebbbd98f5481437a8021afaf58ee7fb1818d347
3
+ size 499723
tokenizer_config.json CHANGED
@@ -5,7 +5,7 @@
5
  "__type": "AddedToken",
6
  "content": "<s>",
7
  "lstrip": false,
8
- "normalized": true,
9
  "rstrip": false,
10
  "single_word": false
11
  },
@@ -14,20 +14,21 @@
14
  "__type": "AddedToken",
15
  "content": "</s>",
16
  "lstrip": false,
17
- "normalized": true,
18
  "rstrip": false,
19
  "single_word": false
20
  },
21
- "legacy": null,
22
  "model_max_length": 1000000000000000019884624838656,
23
  "pad_token": null,
 
24
  "sp_model_kwargs": {},
25
  "tokenizer_class": "LlamaTokenizer",
26
  "unk_token": {
27
  "__type": "AddedToken",
28
  "content": "<unk>",
29
  "lstrip": false,
30
- "normalized": true,
31
  "rstrip": false,
32
  "single_word": false
33
  }
 
5
  "__type": "AddedToken",
6
  "content": "<s>",
7
  "lstrip": false,
8
+ "normalized": false,
9
  "rstrip": false,
10
  "single_word": false
11
  },
 
14
  "__type": "AddedToken",
15
  "content": "</s>",
16
  "lstrip": false,
17
+ "normalized": false,
18
  "rstrip": false,
19
  "single_word": false
20
  },
21
+ "legacy": false,
22
  "model_max_length": 1000000000000000019884624838656,
23
  "pad_token": null,
24
+ "padding_side": "right",
25
  "sp_model_kwargs": {},
26
  "tokenizer_class": "LlamaTokenizer",
27
  "unk_token": {
28
  "__type": "AddedToken",
29
  "content": "<unk>",
30
  "lstrip": false,
31
+ "normalized": false,
32
  "rstrip": false,
33
  "single_word": false
34
  }