MCPcopy Create free account
hub / github.com/encoder-run/operator / process_multiple_files

Method process_multiple_files

cmd/modeldeployer/main.py:31–56  ·  view source on GitHub ↗
(self, file_inputs: List[Dict])

Source from the content-addressed store, hash-verified

29 return self.process_multiple_files(inputs)
30
31 def process_multiple_files(self, file_inputs: List[Dict]) -> Dict:
32 all_chunks = []
33 file_chunk_map = {}
34
35 # Accumulate chunks from all files
36 for file_input in file_inputs:
37 file_path = file_input["file_path"]
38 source_code = file_input["code"]
39 hash = file_input["file_hash"]
40 chunks = self.chunk_code(source_code, hash, 500)
41 all_chunks.extend(chunks)
42 file_chunk_map[file_path] = (len(all_chunks) - len(chunks), len(all_chunks))
43
44 # Encode all chunks at once
45 codes = [chunk["code"] for chunk in all_chunks]
46 code_embs = self.model.encode(codes, convert_to_tensor=True)
47
48 # Distribute embeddings back to respective files
49 results = {}
50 for file_path, (start, end) in file_chunk_map.items():
51 file_chunks = all_chunks[start:end]
52 for chunk, code_emb in zip(file_chunks, code_embs[start:end]):
53 chunk["embedding"] = code_emb.tolist()
54 results[file_path] = {"embeddings": file_chunks}
55
56 return {"results": results}
57
58 def chunk_code(self, code, hash, max_token_length):
59 # Encode the entire code at once, ignoring special tokens

Callers 1

predictMethod · 0.95

Calls 1

chunk_codeMethod · 0.95

Tested by

no test coverage detected