MCPcopy Create free account
hub / github.com/BIT-DataLab/LakeBench / querytable_columns

Function querytable_columns

join&union/InfoGather/union_webtable.py:36–82  ·  view source on GitHub ↗
(tablefile,column,inversed,KIV,KIA,graph,docNo)

Source from the content-addressed store, hash-verified

34
35
36def querytable_columns(tablefile,column,inversed,KIV,KIA,graph,docNo):
37 df = pd.read_csv(tablefile,low_memory=False)
38 columnvalues = df[column].unique()
39
40 columnvalues_len = len(columnvalues)
41
42 KIA_docs_set = set(KIA[column])
43
44 KIV_docs_list = []
45 for value in columnvalues:
46 KIV_docs_list += inversed[value]
47
48 KIV_docs_set = set(KIV_docs_list)
49
50 KIV_KIA_set = KIA_docs_set & KIV_docs_set
51
52 # compute weight
53 weigth_dict = {}
54 for doc in KIV_KIA_set:
55 docvalue_set = set(KIV[doc])
56 doclen = len(docvalue_set)
57 overlapevalues = docvalue_set & set(columnvalues)
58 weight = len(overlapevalues) / min(doclen,columnvalues_len)
59 weigth_dict[doc] = weight
60
61 a=list(KIV_KIA_set)
62 if a is not None and len(a) >= 500:
63 a.sort()
64 a = a[:500] # 对列表a进行切片操作
65 # compute ppr
66 pr = IncrementalPersonalizedPageRank1(graph, 300, 0.3,a)
67 pr.initial_random_walks()
68 pprdict = pr.compute_personalized_page_ranks()
69
70 # get all the weigted dict
71 total_ppr_list = []
72 for doc in a:
73 doc_weigth = weigth_dict[doc]
74 doc_dict_weigth = {key:value*doc_weigth for key,value in pprdict[doc].items()}
75 total_ppr_list.append(doc_dict_weigth)
76
77 # compute the weighted dict
78 total_ppr = defaultdict(float)
79 for ppr_ele in total_ppr_list:
80 for key,value in ppr_ele.items():
81 total_ppr[key] += value
82 return total_ppr
83
84def indicator_union(ki,uninon_path = "/data_ssd/webtable/large/small_query",indexstorepath = "/data/lijiajun/infogather/webtables/index/"):
85

Callers 1

indicator_unionFunction · 0.70

Tested by

no test coverage detected