input: a file path :return: train, test, valid. where train=(trainX, trainY). where trainX: is a list of list.each list representation a sentence.trainY: is a list of label. each label is a number
(vocabulary_word2index,vocabulary_word2index_label,valid_portion=0.05,max_training_data=1000000,training_data_path='train-zhihu4-only-title-all.txt')
| 261 | return train, test, test |
| 262 | |
| 263 | def load_data(vocabulary_word2index,vocabulary_word2index_label,valid_portion=0.05,max_training_data=1000000,training_data_path='train-zhihu4-only-title-all.txt'): # n_words=100000, |
| 264 | """ |
| 265 | input: a file path |
| 266 | :return: train, test, valid. where train=(trainX, trainY). where |
| 267 | trainX: is a list of list.each list representation a sentence.trainY: is a list of label. each label is a number |
| 268 | """ |
| 269 | # 1.load a zhihu data from file |
| 270 | # example:"w305 w6651 w3974 w1005 w54 w109 w110 w3974 w29 w25 w1513 w3645 w6 w111 __label__-400525901828896492" |
| 271 | print("load_data.started...") |
| 272 | zhihu_f = codecs.open(training_data_path, 'r', 'utf8') #-zhihu4-only-title.txt |
| 273 | lines = zhihu_f.readlines() |
| 274 | # 2.transform X as indices |
| 275 | # 3.transform y as scalar |
| 276 | X = [] |
| 277 | Y = [] |
| 278 | for i, line in enumerate(lines): |
| 279 | x, y = line.split('__label__') #x='w17314 w5521 w7729 w767 w10147 w111' |
| 280 | y=y.replace('\n','') |
| 281 | x = x.replace("\t",' EOS ').strip() |
| 282 | if i<5: |
| 283 | print("x0:",x) #get raw x |
| 284 | #x_=process_one_sentence_to_get_ui_bi_tri_gram(x) |
| 285 | #if i<5: |
| 286 | # print("x1:",x_) # |
| 287 | x=x.split(" ") |
| 288 | x = [vocabulary_word2index.get(e,0) for e in x] #if can't find the word, set the index as '0'.(equal to PAD_ID = 0) |
| 289 | if i<5: |
| 290 | print("x1:",x) #word to index |
| 291 | y = vocabulary_word2index_label[y] #np.abs(hash(y)) |
| 292 | X.append(x) |
| 293 | Y.append(y) |
| 294 | # 4.split to train,test and valid data |
| 295 | number_examples = len(X) |
| 296 | print("number_examples:",number_examples) # |
| 297 | train = (X[0:int((1 - valid_portion) * number_examples)], Y[0:int((1 - valid_portion) * number_examples)]) |
| 298 | test = (X[int((1 - valid_portion) * number_examples) + 1:], Y[int((1 - valid_portion) * number_examples) + 1:]) |
| 299 | # 5.return |
| 300 | print("load_data.ended...") |
| 301 | return train, test, test |
| 302 | |
| 303 | # 将一句话转化为(uigram,bigram,trigram)后的字符串 |
| 304 | def process_one_sentence_to_get_ui_bi_tri_gram(sentence,n_gram=3): |
nothing calls this directly
no outgoing calls
no test coverage detected