| 278 | # print(f"提取完成,结果已保存到 {outfile}") |
| 279 | |
| 280 | def get_entities_from_triples(infilename, outfilename, signpass=[]): |
| 281 | cnt = 0 |
| 282 | with open(infilename, 'r', encoding='utf-8') as infile, open(outfilename, 'w', encoding='utf-8') as outfile: |
| 283 | # 初始化前一行的内容 |
| 284 | previous_item = None |
| 285 | for line in infile: |
| 286 | cnt += 1 |
| 287 | # 分割每行,提取第一列 |
| 288 | current_item = line.split("\t")[0] |
| 289 | # 过滤掉特殊符号 |
| 290 | for sign in signpass: |
| 291 | current_item = current_item.strip(sign) |
| 292 | current_item = re.sub(r"_+", " ", current_item) |
| 293 | if current_item != previous_item: |
| 294 | # 写入输出文件 |
| 295 | outfile.write(current_item + '\n') |
| 296 | previous_item = current_item |
| 297 | # 输出处理进度,每处理 10,000 个对象输出一次 |
| 298 | progress = cnt + 1 |
| 299 | if progress % 10000 == 0: |
| 300 | print(f"处理进度: {progress}") |
| 301 | def get_entities_from_wikidata(infilename, outfilename1, outfilename2, outfilename3): |
| 302 | with open(infilename, "r", encoding="utf-8") as infile, open(outfilename1, "w", encoding="utf-8") as outfile1, open(outfilename2, "w", encoding="utf-8") as outfile2, open(outfilename3, "w", encoding="utf-8") as outfile3: |
| 303 | index = 0 |