Tokenizes a piece of text into its word pieces. This uses a greedy longest-match-first algorithm to perform tokenization using the given vocabulary. For example: input = "unaffable" output = ["un", "##aff", "##able"] Args: text: A single token or whitespace separ
(self, text)
| 320 | self.max_input_chars_per_word = max_input_chars_per_word |
| 321 | |
| 322 | def tokenize(self, text): |
| 323 | """Tokenizes a piece of text into its word pieces. |
| 324 | |
| 325 | This uses a greedy longest-match-first algorithm to perform tokenization |
| 326 | using the given vocabulary. |
| 327 | |
| 328 | For example: |
| 329 | input = "unaffable" |
| 330 | output = ["un", "##aff", "##able"] |
| 331 | |
| 332 | Args: |
| 333 | text: A single token or whitespace separated tokens. This should have |
| 334 | already been passed through `BasicTokenizer. |
| 335 | |
| 336 | Returns: |
| 337 | A list of wordpiece tokens. |
| 338 | """ |
| 339 | |
| 340 | text = convert_to_unicode(text) |
| 341 | |
| 342 | output_tokens = [] |
| 343 | for token in whitespace_tokenize(text): |
| 344 | chars = list(token) |
| 345 | if len(chars) > self.max_input_chars_per_word: |
| 346 | output_tokens.append(self.unk_token) |
| 347 | continue |
| 348 | |
| 349 | is_bad = False |
| 350 | start = 0 |
| 351 | sub_tokens = [] |
| 352 | while start < len(chars): |
| 353 | end = len(chars) |
| 354 | cur_substr = None |
| 355 | while start < end: |
| 356 | substr = "".join(chars[start:end]) |
| 357 | if start > 0: |
| 358 | substr = "##" + substr |
| 359 | if substr in self.vocab: |
| 360 | cur_substr = substr |
| 361 | break |
| 362 | end -= 1 |
| 363 | if cur_substr is None: |
| 364 | is_bad = True |
| 365 | break |
| 366 | sub_tokens.append(cur_substr) |
| 367 | start = end |
| 368 | |
| 369 | if is_bad: |
| 370 | output_tokens.append(self.unk_token) |
| 371 | else: |
| 372 | output_tokens.extend(sub_tokens) |
| 373 | return output_tokens |
| 374 | |
| 375 | |
| 376 | def _is_whitespace(char): |