(id, values)
| 447 | return results |
| 448 | |
| 449 | def pre_process_vector(id, values): |
| 450 | corrected_values = [float(v) for v in values] |
| 451 | return { |
| 452 | "id": str(id), # Keep as string for server compatibility |
| 453 | "dense_values": corrected_values, |
| 454 | "document_id": f"doc_{id//10}" # Group vectors into documents |
| 455 | } |
| 456 | |
| 457 | def read_dataset_from_parquet(dataset_name, max_vectors=50000): |
| 458 | """Read dataset from parquet files with a strict limit to prevent memory issues.""" |
no outgoing calls
no test coverage detected