MCPcopy Create free account
hub / github.com/BIT-DataLab/LakeBench / __init__

Method __init__

join/Deepjoin/hnsw_search.py:12–35  ·  view source on GitHub ↗
(self,table_path,index_path,scale)

Source from the content-addressed store, hash-verified

10
11class HNSWSearcher(object):
12 def __init__(self,table_path,index_path,scale):
13 tfile = open(table_path,"rb")
14 tables = pickle.load(tfile)
15 # For scalability experiments: load a percentage of tables
16 self.tables = random.sample(tables, int(scale*len(tables)))
17 print("From %d total data-lake tables, scale down to %d tables" % (len(tables), len(self.tables)))
18 tfile.close()
19 self.vec_dim = len(self.tables[1][1][0])
20
21 index_start_time = time.time()
22 self.index = hnswlib.Index(space='cosine', dim=self.vec_dim)
23 self.all_columns, self.col_table_ids = self._preprocess_table_hnsw()
24 # if not os.path.exists(index_path):
25 # build index from scratch
26 # self.index.init_index(max_elements=len(self.all_columns), ef_construction=100, M=16)
27 self.index.init_index(max_elements=len(self.all_columns), ef_construction=100, M=32)
28
29 self.index.set_ef(10)
30 self.index.add_items(self.all_columns)
31 # self.index.save_index(index_path)
32 print("--- Indexing Time: %s seconds ---" % (time.time() - index_start_time))
33 # else:
34 # # load index
35 # self.index.load_index(index_path, max_elements = len(self.all_columns))
36
37 def topk(self, enc, query, K, N=5, threshold=0.6):
38 # Note: N is the number of columns retrieved from the index

Callers

nothing calls this directly

Calls 2

closeMethod · 0.80

Tested by

no test coverage detected