(self, data_dir, header=False, drop_keyword=False)
| 358 | """Processor for Semeval Task9 data set.""" |
| 359 | |
| 360 | def get_train_examples(self, data_dir, header=False, drop_keyword=False): |
| 361 | lines = self._read_csv(data_dir + '/V1.4_Training.csv') |
| 362 | examples = [] |
| 363 | if drop_keyword: |
| 364 | keywords = [ |
| 365 | line.strip() for line in open(data_dir + '/../keywords') |
| 366 | ] |
| 367 | |
| 368 | for i, line in enumerate(lines): |
| 369 | if i == 0 and header: |
| 370 | continue |
| 371 | guid = line[0] |
| 372 | text_a = tokenization.convert_to_unicode(line[1]) |
| 373 | text_a = clean_str(text_a) |
| 374 | |
| 375 | if drop_keyword: |
| 376 | new_tokens = [] |
| 377 | for w in text_a.split(' '): |
| 378 | if w in keywords and random.random() > 0.8: |
| 379 | continue |
| 380 | new_tokens.append(w) |
| 381 | text_a = ' '.join(new_tokens) |
| 382 | text_b = None |
| 383 | label = line[2] |
| 384 | examples.append( |
| 385 | InputExample( |
| 386 | guid=guid, text_a=text_a, text_b=text_b, label=label)) |
| 387 | return examples |
| 388 | |
| 389 | def get_dev_examples(self, data_dir, header=True): |
| 390 | lines = self._read_csv(data_dir + '/SubtaskA_Trial_Test_Labeled.csv') |
nothing calls this directly
no test coverage detected