Converts a single `InputExample` into a single `InputFeatures`.
(ex_index, example, label_list, max_seq_length,
tokenizer)
| 303 | |
| 304 | |
| 305 | def convert_single_example(ex_index, example, label_list, max_seq_length, |
| 306 | tokenizer): |
| 307 | """Converts a single `InputExample` into a single `InputFeatures`.""" |
| 308 | |
| 309 | if isinstance(example, PaddingInputExample): |
| 310 | return InputFeatures( |
| 311 | input_ids=[0] * max_seq_length, |
| 312 | input_mask=[0] * max_seq_length, |
| 313 | segment_ids=[0] * max_seq_length, |
| 314 | label_id=0, |
| 315 | is_real_example=False) |
| 316 | |
| 317 | label_map = {} |
| 318 | for (i, label) in enumerate(label_list): |
| 319 | label_map[label] = i |
| 320 | |
| 321 | tokens_a = tokenizer.tokenize(example.text_a) |
| 322 | tokens_b = None |
| 323 | if example.text_b: |
| 324 | tokens_b = tokenizer.tokenize(example.text_b) |
| 325 | |
| 326 | if tokens_b: |
| 327 | # Modifies `tokens_a` and `tokens_b` in place so that the total |
| 328 | # length is less than the specified length. |
| 329 | # Account for [CLS], [SEP], [SEP] with "- 3" |
| 330 | _truncate_seq_pair(tokens_a, tokens_b, max_seq_length - 3) |
| 331 | else: |
| 332 | # Account for [CLS] and [SEP] with "- 2" |
| 333 | if len(tokens_a) > max_seq_length - 2: |
| 334 | tokens_a = tokens_a[0:(max_seq_length - 2)] |
| 335 | |
| 336 | # The convention in BERT is: |
| 337 | # (a) For sequence pairs: |
| 338 | # tokens: [CLS] is this jack ##son ##ville ? [SEP] no it is not . [SEP] |
| 339 | # type_ids: 0 0 0 0 0 0 0 0 1 1 1 1 1 1 |
| 340 | # (b) For single sequences: |
| 341 | # tokens: [CLS] the dog is hairy . [SEP] |
| 342 | # type_ids: 0 0 0 0 0 0 0 |
| 343 | # |
| 344 | # Where "type_ids" are used to indicate whether this is the first |
| 345 | # sequence or the second sequence. The embedding vectors for `type=0` and |
| 346 | # `type=1` were learned during pre-training and are added to the wordpiece |
| 347 | # embedding vector (and position vector). This is not *strictly* necessary |
| 348 | # since the [SEP] token unambiguously separates the sequences, but it makes |
| 349 | # it easier for the model to learn the concept of sequences. |
| 350 | # |
| 351 | # For classification tasks, the first vector (corresponding to [CLS]) is |
| 352 | # used as the "sentence vector". Note that this only makes sense because |
| 353 | # the entire model is fine-tuned. |
| 354 | tokens = [] |
| 355 | segment_ids = [] |
| 356 | tokens.append("[CLS]") |
| 357 | segment_ids.append(0) |
| 358 | for token in tokens_a: |
| 359 | tokens.append(token) |
| 360 | segment_ids.append(0) |
| 361 | tokens.append("[SEP]") |
| 362 | segment_ids.append(0) |
no test coverage detected