(tablefile,column,inversed,KIV,KIA,graph,docNo)
| 34 | |
| 35 | |
| 36 | def querytable_columns(tablefile,column,inversed,KIV,KIA,graph,docNo): |
| 37 | df = pd.read_csv(tablefile,low_memory=False) |
| 38 | columnvalues = df[column].unique() |
| 39 | |
| 40 | columnvalues_len = len(columnvalues) |
| 41 | |
| 42 | KIA_docs_set = set(KIA[column]) |
| 43 | |
| 44 | KIV_docs_list = [] |
| 45 | for value in columnvalues: |
| 46 | KIV_docs_list += inversed[value] |
| 47 | |
| 48 | KIV_docs_set = set(KIV_docs_list) |
| 49 | |
| 50 | KIV_KIA_set = KIA_docs_set & KIV_docs_set |
| 51 | |
| 52 | # compute weight |
| 53 | weigth_dict = {} |
| 54 | for doc in KIV_KIA_set: |
| 55 | docvalue_set = set(KIV[doc]) |
| 56 | doclen = len(docvalue_set) |
| 57 | overlapevalues = docvalue_set & set(columnvalues) |
| 58 | weight = len(overlapevalues) / min(doclen,columnvalues_len) |
| 59 | weigth_dict[doc] = weight |
| 60 | |
| 61 | a=list(KIV_KIA_set) |
| 62 | if a is not None and len(a) >= 500: |
| 63 | a.sort() |
| 64 | a = a[:500] # 对列表a进行切片操作 |
| 65 | # compute ppr |
| 66 | pr = IncrementalPersonalizedPageRank1(graph, 300, 0.3,a) |
| 67 | pr.initial_random_walks() |
| 68 | pprdict = pr.compute_personalized_page_ranks() |
| 69 | |
| 70 | # get all the weigted dict |
| 71 | total_ppr_list = [] |
| 72 | for doc in a: |
| 73 | doc_weigth = weigth_dict[doc] |
| 74 | doc_dict_weigth = {key:value*doc_weigth for key,value in pprdict[doc].items()} |
| 75 | total_ppr_list.append(doc_dict_weigth) |
| 76 | |
| 77 | # compute the weighted dict |
| 78 | total_ppr = defaultdict(float) |
| 79 | for ppr_ele in total_ppr_list: |
| 80 | for key,value in ppr_ele.items(): |
| 81 | total_ppr[key] += value |
| 82 | return total_ppr |
| 83 | |
| 84 | def indicator_union(ki,uninon_path = "/data_ssd/webtable/large/small_query",indexstorepath = "/data/lijiajun/infogather/webtables/index/"): |
| 85 |
no test coverage detected