(qpath: str,
save_root: str,
result_root: str,
k:int)
| 10 | from heap import * |
| 11 | |
| 12 | def api(qpath: str, |
| 13 | save_root: str, |
| 14 | result_root: str, |
| 15 | k:int): |
| 16 | |
| 17 | # 读取文件 |
| 18 | with open("results_web/setMap.pkl", "rb") as tf: |
| 19 | setMap = pickle.load(tf) |
| 20 | outpath = os.path.join(save_root, "outputs") |
| 21 | tf = open(outpath + "integerSet.json", "r") |
| 22 | integerSet = json.load(tf) |
| 23 | tf = open(outpath + "PLs.json", "r") |
| 24 | PLs = json.load(tf) |
| 25 | print("PLs") |
| 26 | tf = open(outpath+"rawDict.json", "r") |
| 27 | rawDict= json.load(tf) |
| 28 | print("rawDict") |
| 29 | tf.close() |
| 30 | table_names = os.listdir(qpath) |
| 31 | |
| 32 | print("load suc!") |
| 33 | |
| 34 | durs = [] |
| 35 | res=[] |
| 36 | i=0 |
| 37 | for table_name in tqdm(table_names): |
| 38 | i+=1 |
| 39 | # try: |
| 40 | ignore=False |
| 41 | query_ID = -1 |
| 42 | table_path = os.path.join(qpath, table_name) |
| 43 | df = pd.read_csv(table_path) |
| 44 | for column_name in df.columns: |
| 45 | query_ID=readQueryID(setMap,table_name,column_name) |
| 46 | if query_ID>0: |
| 47 | ignore = True |
| 48 | raw_tokens = list(set(df[column_name].tolist())) |
| 49 | t1 = time.time() |
| 50 | result=searchMergeProbeCostModelGreedy(integerSet, PLs, raw_tokens,rawDict,setMap, k, ignore,query_ID) |
| 51 | print(result) |
| 52 | t2 = time.time() |
| 53 | if result!=0: |
| 54 | dur = (t2 - t1) |
| 55 | durs.append(dur) |
| 56 | # print(f"在线处理一个query时间:{dur:.2f}秒,query长度:{len(raw_tokens)}") |
| 57 | for x in result: |
| 58 | # print(table_name,setMap[x+1]["table_name"],column_name,setMap[x+1]["column_name"]) |
| 59 | re=[table_name,setMap[x+1]["table_name"],column_name,setMap[x+1]["column_name"]] |
| 60 | res.append(re) |
| 61 | mean = np.mean(durs) |
| 62 | print(f"平均时间:{mean:.2f}秒") |
| 63 | df_end = pd.DataFrame(res, columns=['query_table','candidate_table','query_column','candidate_column']) |
| 64 | result_path=os.path.join(result_root, "join_top"+k+".csv") |
| 65 | df_end.to_csv(result_path, index=False) |
nothing calls this directly
no test coverage detected