`IMDB `_ is a Movie Review Sentiment Classification dataset. we use dataset provided by `LOTClass `_
| 148 | |
| 149 | |
| 150 | class ImdbProcessor(DataProcessor): |
| 151 | """ |
| 152 | `IMDB <https://ai.stanford.edu/~ang/papers/acl11-WordVectorsSentimentAnalysis.pdf>`_ is a Movie Review Sentiment Classification dataset. |
| 153 | |
| 154 | we use dataset provided by `LOTClass <https://github.com/yumeng5/LOTClass>`_ |
| 155 | """ |
| 156 | |
| 157 | def __init__(self): |
| 158 | super().__init__() |
| 159 | self.labels = ["negative", "positive"] |
| 160 | |
| 161 | def get_examples(self, data_dir, split): |
| 162 | examples = [] |
| 163 | label_file = open(os.path.join(data_dir, "{}_labels.txt".format(split)), 'r') |
| 164 | labels = [int(x.strip()) for x in label_file.readlines()] |
| 165 | with open(os.path.join(data_dir, '{}.txt'.format(split)),'r') as fin: |
| 166 | for idx, line in enumerate(fin): |
| 167 | text_a = line.strip() |
| 168 | example = InputExample(guid=str(idx), text_a=text_a, label=int(labels[idx])) |
| 169 | examples.append(example) |
| 170 | return examples |
| 171 | |
| 172 | |
| 173 | @staticmethod |
| 174 | def get_test_labels_only(data_dir, dirname): |
| 175 | label_file = open(os.path.join(data_dir,dirname,"{}_labels.txt".format('test')),'r') |
| 176 | labels = [int(x.strip()) for x in label_file.readlines()] |
| 177 | return labels |
| 178 | |
| 179 | class YahooProcessor(DataProcessor): |
| 180 | """ |
nothing calls this directly
no outgoing calls
no test coverage detected