from dataclasses import dataclass from typing import Optional import numpy as np @dataclass class EntityCandidate: raw_name: str normalized_name: str language: str doc_id: str embedding: Optional[np.ndarray] = None def resolve_against_live_graph( candidate: EntityCandidate, graph_client, # neo4j.GraphDatabase or similar embed_fn, threshold: float = 0.85, ) -> Optional[str]: """ Run entity resolution against existing graph nodes before inserting. Returns the existing node ID if a match is found, else None. """ # Step 1: Blocking — retrieve candidate neighbors by name prefix blocking_query = """ MATCH (e:Entity) WHERE e.blocking_key STARTS WITH $prefix RETURN e.id AS id, e.canonical_name AS name, e.name_embedding AS emb LIMIT 50 """ prefix = candidate.normalized_name[:4].upper() neighbors = graph_client.run(blocking_query, prefix=prefix).data() if not neighbors: return None # Step 2: Embedding similarity against blocked candidates if candidate.embedding is None: candidate.embedding = embed_fn(candidate.normalized_name) best_score = 0.0 best_id = None for row in neighbors: if row["emb"] is None: continue existing_emb = np.array(row["emb"]) score = float(np.dot(candidate.embedding, existing_emb) / (np.linalg.norm(candidate.embedding) * np.linalg.norm(existing_emb) + 1e-9)) if score > best_score: best_score = score best_id = row["id"] return best_id if best_score >= threshold else None