MCPcopy Create free account
hub / github.com/THUDM/GLM / __init__

Method __init__

data_utils/wordpiece.py:77–105  ·  view source on GitHub ↗

Constructs a BertTokenizer. Args: vocab_file: Path to a one-wordpiece-per-line vocabulary file do_lower_case: Whether to lower case the input Only has an effect when do_wordpiece_only=False do_basic_tokenize: Whether to do basic tokeniz

(self, vocab_file, do_lower_case=True, max_len=None, do_basic_tokenize=True,
                 never_split=("[UNK]", "[SEP]", "[PAD]", "[CLS]", "[MASK]"))

Source from the content-addressed store, hash-verified

75 """Runs end-to-end tokenization: punctuation splitting + wordpiece"""
76
77 def __init__(self, vocab_file, do_lower_case=True, max_len=None, do_basic_tokenize=True,
78 never_split=("[UNK]", "[SEP]", "[PAD]", "[CLS]", "[MASK]")):
79 """Constructs a BertTokenizer.
80
81 Args:
82 vocab_file: Path to a one-wordpiece-per-line vocabulary file
83 do_lower_case: Whether to lower case the input
84 Only has an effect when do_wordpiece_only=False
85 do_basic_tokenize: Whether to do basic tokenization before wordpiece.
86 max_len: An artificial maximum length to truncate tokenized sequences to;
87 Effective maximum length is always the minimum of this
88 value (if specified) and the underlying BERT model's
89 sequence length.
90 never_split: List of tokens which will never be split during tokenization.
91 Only has an effect when do_wordpiece_only=False
92 """
93 if not os.path.isfile(vocab_file):
94 raise ValueError(
95 "Can't find a vocabulary file at path '{}'. To load the vocabulary from a Google pretrained "
96 "model use `tokenizer = BertTokenizer.from_pretrained(PRETRAINED_MODEL_NAME)`".format(vocab_file))
97 self.vocab = load_vocab(vocab_file)
98 self.ids_to_tokens = collections.OrderedDict(
99 [(ids, tok) for tok, ids in self.vocab.items()])
100 self.do_basic_tokenize = do_basic_tokenize
101 if do_basic_tokenize:
102 self.basic_tokenizer = BasicTokenizer(do_lower_case=do_lower_case,
103 never_split=never_split)
104 self.wordpiece_tokenizer = WordpieceTokenizer(vocab=self.vocab)
105 self.max_len = max_len if max_len is not None else int(1e12)
106
107 def tokenize(self, text):
108 if self.do_basic_tokenize:

Callers

nothing calls this directly

Calls 3

load_vocabFunction · 0.85
BasicTokenizerClass · 0.85
WordpieceTokenizerClass · 0.85

Tested by

no test coverage detected