-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathgraph_diagnostic.py
More file actions
233 lines (200 loc) · 7.95 KB
/
Copy pathgraph_diagnostic.py
File metadata and controls
233 lines (200 loc) · 7.95 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
"""
Database Diagnostic Tool - Inspect Neo4j Graph Contents
"""
from neo4j import GraphDatabase
from loguru import logger
class GraphDiagnostic:
"""Diagnose graph database contents"""
def __init__(self, uri="bolt://localhost:7687", user="neo4j", password="password123"):
self.driver = GraphDatabase.driver(uri, auth=(user, password))
logger.info(f"Connected to Neo4j at {uri}")
def close(self):
self.driver.close()
def get_all_nodes_summary(self):
"""Get summary of all nodes by type"""
query = """
MATCH (n)
RETURN labels(n)[0] as type, count(n) as count
ORDER BY count DESC
"""
with self.driver.session() as session:
result = session.run(query)
return [{"type": r["type"], "count": r["count"]} for r in result]
def get_all_relationships_summary(self):
"""Get summary of all relationships by type"""
query = """
MATCH ()-[r]->()
RETURN type(r) as type, count(r) as count
ORDER BY count DESC
"""
with self.driver.session() as session:
result = session.run(query)
return [{"type": r["type"], "count": r["count"]} for r in result]
def find_entity_by_name(self, name):
"""Find all entities with given name"""
query = """
MATCH (n)
WHERE toLower(n.name) = toLower($name)
RETURN n.name as name, labels(n) as labels, id(n) as id
"""
with self.driver.session() as session:
result = session.run(query, name=name)
return [{"name": r["name"], "labels": r["labels"], "id": r["id"]} for r in result]
def get_sample_nodes(self, node_type, limit=10):
"""Get sample nodes of a specific type"""
query = f"""
MATCH (n:{node_type})
RETURN n.name as name, id(n) as id
LIMIT $limit
"""
with self.driver.session() as session:
result = session.run(query, limit=limit)
return [{"name": r["name"], "id": r["id"]} for r in result]
def get_entity_connections(self, name):
"""Get all connections for an entity"""
query = """
MATCH (e {name: $name})
OPTIONAL MATCH (e)-[r_out]->(target)
OPTIONAL MATCH (source)-[r_in]->(e)
RETURN
labels(e)[0] as entity_type,
collect(DISTINCT {
type: type(r_out),
target: target.name,
target_type: labels(target)[0]
}) as outgoing,
collect(DISTINCT {
type: type(r_in),
source: source.name,
source_type: labels(source)[0]
}) as incoming
"""
with self.driver.session() as session:
result = session.run(query, name=name)
record = result.single()
if not record:
return None
return {
"entity_type": record["entity_type"],
"outgoing": [r for r in record["outgoing"] if r["target"]],
"incoming": [r for r in record["incoming"] if r["source"]]
}
def find_duplicates(self):
"""Find entities with same name but different types"""
query = """
MATCH (n)
WITH n.name as name, collect(DISTINCT labels(n)[0]) as types, count(*) as count
WHERE count > 1
RETURN name, types, count
ORDER BY count DESC
"""
with self.driver.session() as session:
result = session.run(query)
return [{"name": r["name"], "types": r["types"], "count": r["count"]} for r in result]
def search_similar_names(self, name):
"""Find entities with similar names"""
query = """
MATCH (n)
WHERE toLower(n.name) CONTAINS toLower($name)
RETURN n.name as name, labels(n)[0] as type
ORDER BY n.name
LIMIT 20
"""
with self.driver.session() as session:
result = session.run(query, name=name)
return [{"name": r["name"], "type": r["type"]} for r in result]
def run_diagnostics():
"""Run comprehensive diagnostics"""
print("\n" + "="*80)
print("GRAPH DATABASE DIAGNOSTICS")
print("="*80)
diag = GraphDiagnostic()
# 1. Node summary
print("\n[1] NODE SUMMARY")
print("-" * 80)
node_summary = diag.get_all_nodes_summary()
total_nodes = sum(n["count"] for n in node_summary)
print(f"Total nodes: {total_nodes}\n")
for node in node_summary:
print(f" {node['type']:<20} {node['count']:>5} nodes")
# 2. Relationship summary
print("\n[2] RELATIONSHIP SUMMARY")
print("-" * 80)
rel_summary = diag.get_all_relationships_summary()
total_rels = sum(r["count"] for r in rel_summary)
print(f"Total relationships: {total_rels}\n")
for rel in rel_summary:
print(f" {rel['type']:<20} {rel['count']:>5} relationships")
# 3. Check for duplicates
print("\n[3] DUPLICATE ENTITIES (same name, different types)")
print("-" * 80)
duplicates = diag.find_duplicates()
if duplicates:
for dup in duplicates:
print(f" '{dup['name']}' appears {dup['count']}x as: {', '.join(dup['types'])}")
else:
print(" No duplicates found")
# 4. Search for key entities
print("\n[4] KEY ENTITIES CHECK")
print("-" * 80)
key_entities = ["Helsing", "Maria Weber", "ARX Robotics", "Munich", "NATO"]
for entity_name in key_entities:
print(f"\n Searching for: '{entity_name}'")
matches = diag.find_entity_by_name(entity_name)
if matches:
for match in matches:
print(f" [OK] Found: {match['name']} (Type: {match['labels'][0]}, ID: {match['id']})")
# Get connections
connections = diag.get_entity_connections(match['name'])
if connections:
out_count = len(connections['outgoing'])
in_count = len(connections['incoming'])
print(f" Connections: {out_count} outgoing, {in_count} incoming")
else:
print(f" ! Not found!")
# Search for similar
similar = diag.search_similar_names(entity_name)
if similar:
print(f" Similar names found:")
for sim in similar[:5]:
print(f" - {sim['name']} ({sim['type']})")
# 5. Sample Organizations
print("\n[5] SAMPLE ORGANIZATIONS (first 10)")
print("-" * 80)
orgs = diag.get_sample_nodes("Organization", limit=10)
for org in orgs:
print(f" - {org['name']} (ID: {org['id']})")
# 6. Sample People
print("\n[6] SAMPLE PEOPLE (first 10)")
print("-" * 80)
people = diag.get_sample_nodes("Person", limit=10)
for person in people:
print(f" - {person['name']} (ID: {person['id']})")
# 7. Sample Locations
print("\n[7] SAMPLE LOCATIONS (first 10)")
print("-" * 80)
locations = diag.get_sample_nodes("Location", limit=10)
for loc in locations:
print(f" - {loc['name']} (ID: {loc['id']})")
diag.close()
print("\n" + "="*80)
print("[OK] Diagnostics completed!")
print("="*80)
# Recommendations
print("\nRECOMMENDATIONS:")
print("-" * 80)
if duplicates:
print("! DUPLICATES FOUND - Consider cleaning the database")
print(" Run: MATCH (n) WHERE n.name = 'EntityName' DELETE n")
else:
print("[OK] No duplicate entities")
if total_rels < 10:
print("! LOW RELATIONSHIP COUNT - Relation extraction may need improvement")
else:
print(f"[OK] Good relationship count ({total_rels})")
print("\nNEXT STEPS:")
print("1. Check if key entities exist with correct names")
print("2. Fix any duplicates or misclassified entities")
print("3. Re-run query engine tests")
if __name__ == "__main__":
run_diagnostics()