Constructs a BertTokenizer. Args: vocab_file: Path to a one-wordpiece-per-line vocabulary file do_lower_case: Whether to lower case the input Only has an effect when do_wordpiece_only=False do_basic_tokenize: Whether to do basic tokeniz
(self, vocab_file, do_lower_case=True, max_len=None, do_basic_tokenize=True,
never_split=("[UNK]", "[SEP]", "[PAD]", "[CLS]", "[MASK]"))
| 75 | """Runs end-to-end tokenization: punctuation splitting + wordpiece""" |
| 76 | |
| 77 | def __init__(self, vocab_file, do_lower_case=True, max_len=None, do_basic_tokenize=True, |
| 78 | never_split=("[UNK]", "[SEP]", "[PAD]", "[CLS]", "[MASK]")): |
| 79 | """Constructs a BertTokenizer. |
| 80 | |
| 81 | Args: |
| 82 | vocab_file: Path to a one-wordpiece-per-line vocabulary file |
| 83 | do_lower_case: Whether to lower case the input |
| 84 | Only has an effect when do_wordpiece_only=False |
| 85 | do_basic_tokenize: Whether to do basic tokenization before wordpiece. |
| 86 | max_len: An artificial maximum length to truncate tokenized sequences to; |
| 87 | Effective maximum length is always the minimum of this |
| 88 | value (if specified) and the underlying BERT model's |
| 89 | sequence length. |
| 90 | never_split: List of tokens which will never be split during tokenization. |
| 91 | Only has an effect when do_wordpiece_only=False |
| 92 | """ |
| 93 | if not os.path.isfile(vocab_file): |
| 94 | raise ValueError( |
| 95 | "Can't find a vocabulary file at path '{}'. To load the vocabulary from a Google pretrained " |
| 96 | "model use `tokenizer = BertTokenizer.from_pretrained(PRETRAINED_MODEL_NAME)`".format(vocab_file)) |
| 97 | self.vocab = load_vocab(vocab_file) |
| 98 | self.ids_to_tokens = collections.OrderedDict( |
| 99 | [(ids, tok) for tok, ids in self.vocab.items()]) |
| 100 | self.do_basic_tokenize = do_basic_tokenize |
| 101 | if do_basic_tokenize: |
| 102 | self.basic_tokenizer = BasicTokenizer(do_lower_case=do_lower_case, |
| 103 | never_split=never_split) |
| 104 | self.wordpiece_tokenizer = WordpieceTokenizer(vocab=self.vocab) |
| 105 | self.max_len = max_len if max_len is not None else int(1e12) |
| 106 | |
| 107 | def tokenize(self, text): |
| 108 | if self.do_basic_tokenize: |
nothing calls this directly
no test coverage detected