MCPcopy Create free account
hub / github.com/brightmart/text_classification / load_data

Function load_data

a08_EntityNetwork/data_util_zhihu.py:263–301  ·  view source on GitHub ↗

input: a file path :return: train, test, valid. where train=(trainX, trainY). where trainX: is a list of list.each list representation a sentence.trainY: is a list of label. each label is a number

(vocabulary_word2index,vocabulary_word2index_label,valid_portion=0.05,max_training_data=1000000,training_data_path='train-zhihu4-only-title-all.txt')

Source from the content-addressed store, hash-verified

261 return train, test, test
262
263def load_data(vocabulary_word2index,vocabulary_word2index_label,valid_portion=0.05,max_training_data=1000000,training_data_path='train-zhihu4-only-title-all.txt'): # n_words=100000,
264 """
265 input: a file path
266 :return: train, test, valid. where train=(trainX, trainY). where
267 trainX: is a list of list.each list representation a sentence.trainY: is a list of label. each label is a number
268 """
269 # 1.load a zhihu data from file
270 # example:"w305 w6651 w3974 w1005 w54 w109 w110 w3974 w29 w25 w1513 w3645 w6 w111 __label__-400525901828896492"
271 print("load_data.started...")
272 zhihu_f = codecs.open(training_data_path, 'r', 'utf8') #-zhihu4-only-title.txt
273 lines = zhihu_f.readlines()
274 # 2.transform X as indices
275 # 3.transform y as scalar
276 X = []
277 Y = []
278 for i, line in enumerate(lines):
279 x, y = line.split('__label__') #x='w17314 w5521 w7729 w767 w10147 w111'
280 y=y.replace('\n','')
281 x = x.replace("\t",' EOS ').strip()
282 if i<5:
283 print("x0:",x) #get raw x
284 #x_=process_one_sentence_to_get_ui_bi_tri_gram(x)
285 #if i<5:
286 # print("x1:",x_) #
287 x=x.split(" ")
288 x = [vocabulary_word2index.get(e,0) for e in x] #if can't find the word, set the index as '0'.(equal to PAD_ID = 0)
289 if i<5:
290 print("x1:",x) #word to index
291 y = vocabulary_word2index_label[y] #np.abs(hash(y))
292 X.append(x)
293 Y.append(y)
294 # 4.split to train,test and valid data
295 number_examples = len(X)
296 print("number_examples:",number_examples) #
297 train = (X[0:int((1 - valid_portion) * number_examples)], Y[0:int((1 - valid_portion) * number_examples)])
298 test = (X[int((1 - valid_portion) * number_examples) + 1:], Y[int((1 - valid_portion) * number_examples) + 1:])
299 # 5.return
300 print("load_data.ended...")
301 return train, test, test
302
303 # 将一句话转化为(uigram,bigram,trigram)后的字符串
304def process_one_sentence_to_get_ui_bi_tri_gram(sentence,n_gram=3):

Callers

nothing calls this directly

Calls

no outgoing calls

Tested by

no test coverage detected