Deduplicate near-identical entities in a knowledge graph. Args: nodes: list of node dicts with at minimum {"id": str, "label": str} edges: list of edge dicts with {"source": str, "target": str, ...} communities: mapping of node_id -> community_id (from cluster())
(
nodes: list[dict],
edges: list[dict],
*,
communities: dict[str, int],
dedup_llm_backend: str | None = None,
)
| 190 | # ── main entry point ────────────────────────────────────────────────────────── |
| 191 | |
| 192 | def deduplicate_entities( |
| 193 | nodes: list[dict], |
| 194 | edges: list[dict], |
| 195 | *, |
| 196 | communities: dict[str, int], |
| 197 | dedup_llm_backend: str | None = None, |
| 198 | ) -> tuple[list[dict], list[dict]]: |
| 199 | """Deduplicate near-identical entities in a knowledge graph. |
| 200 | |
| 201 | Args: |
| 202 | nodes: list of node dicts with at minimum {"id": str, "label": str} |
| 203 | edges: list of edge dicts with {"source": str, "target": str, ...} |
| 204 | communities: mapping of node_id -> community_id (from cluster()) |
| 205 | dedup_llm_backend: if set, use LLM to resolve ambiguous pairs |
| 206 | |
| 207 | Returns: |
| 208 | (deduped_nodes, deduped_edges) with edges rewired to survivors |
| 209 | """ |
| 210 | # Guard: cross-project dedup is not supported — nodes from different repos |
| 211 | # share label names by coincidence and must never be merged by string similarity. |
| 212 | # If you need to dedup a global graph, run deduplicate_entities per-repo first. |
| 213 | repos_seen = {n.get("repo") for n in nodes if n.get("repo")} |
| 214 | if len(repos_seen) > 1: |
| 215 | raise ValueError( |
| 216 | f"deduplicate_entities: nodes span multiple repos {sorted(repos_seen)!r}. " |
| 217 | f"Cross-project dedup is disabled — run dedup per-repo before merging." |
| 218 | ) |
| 219 | |
| 220 | if len(nodes) <= 1: |
| 221 | return nodes, edges |
| 222 | |
| 223 | # Pre-deduplicate: keep first occurrence of each id. |
| 224 | # Warn when two nodes share an ID but originate from different source files — |
| 225 | # this indicates a cross-chunk ID collision (#1504) where silent data loss occurs. |
| 226 | seen_ids: dict[str, dict] = {} |
| 227 | for node in nodes: |
| 228 | nid = node.get("id", "") |
| 229 | if not nid: |
| 230 | continue |
| 231 | if nid not in seen_ids: |
| 232 | seen_ids[nid] = node |
| 233 | else: |
| 234 | existing_sf = seen_ids[nid].get("source_file") or "" |
| 235 | new_sf = node.get("source_file") or "" |
| 236 | if existing_sf != new_sf: |
| 237 | print( |
| 238 | f"[graphify] WARNING: node '{nid}' from '{new_sf}' collides with " |
| 239 | f"node from '{existing_sf}' — the second node will be dropped. " |
| 240 | f"This is a cross-chunk ID collision caused by two files with the " |
| 241 | f"same name in different directories. To avoid data loss, run " |
| 242 | f"'graphify extract' per subfolder and merge with " |
| 243 | f"'graphify merge-graphs'.", |
| 244 | file=sys.stderr, |
| 245 | ) |
| 246 | unique_nodes = list(seen_ids.values()) |
| 247 | |
| 248 | if len(unique_nodes) <= 1: |
| 249 | return unique_nodes, edges |