codegraph-engine 2.1.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. codegraph/__init__.py +37 -0
  2. codegraph/agent.py +26 -0
  3. codegraph/architecture.py +328 -0
  4. codegraph/audit.py +106 -0
  5. codegraph/cache.py +95 -0
  6. codegraph/cli.py +854 -0
  7. codegraph/config.py +43 -0
  8. codegraph/constraints.py +238 -0
  9. codegraph/context.py +1228 -0
  10. codegraph/epistemic.py +90 -0
  11. codegraph/errors.py +275 -0
  12. codegraph/evidence/__init__.py +15 -0
  13. codegraph/evidence/citations.py +397 -0
  14. codegraph/frameworks.py +434 -0
  15. codegraph/freshness.py +295 -0
  16. codegraph/git.py +278 -0
  17. codegraph/graph/__init__.py +46 -0
  18. codegraph/graph/models.py +41 -0
  19. codegraph/graph/traversal.py +1291 -0
  20. codegraph/indexing/__init__.py +4 -0
  21. codegraph/indexing/classifier.py +274 -0
  22. codegraph/indexing/indexer.py +943 -0
  23. codegraph/indexing/models.py +338 -0
  24. codegraph/indexing/parser.py +1240 -0
  25. codegraph/indexing/scanner.py +200 -0
  26. codegraph/indexing/test_framework.py +116 -0
  27. codegraph/interrogation.py +1582 -0
  28. codegraph/llm/__init__.py +3 -0
  29. codegraph/llm/base.py +15 -0
  30. codegraph/llm/context.py +20 -0
  31. codegraph/mcp/__init__.py +3 -0
  32. codegraph/mcp/server.py +736 -0
  33. codegraph/memory/__init__.py +3 -0
  34. codegraph/memory/store.py +46 -0
  35. codegraph/models.py +289 -0
  36. codegraph/observability.py +151 -0
  37. codegraph/optimizer.py +372 -0
  38. codegraph/planner.py +417 -0
  39. codegraph/py.typed +1 -0
  40. codegraph/query_expansion.py +199 -0
  41. codegraph/ranking.py +363 -0
  42. codegraph/resolver.py +843 -0
  43. codegraph/resources/__init__.py +45 -0
  44. codegraph/resources/cache.py +117 -0
  45. codegraph/resources/coalescer.py +83 -0
  46. codegraph/resources/debouncer.py +98 -0
  47. codegraph/resources/governor.py +232 -0
  48. codegraph/resources/policy.py +123 -0
  49. codegraph/retrieval_policy.py +220 -0
  50. codegraph/search/__init__.py +23 -0
  51. codegraph/search/hybrid.py +301 -0
  52. codegraph/search/semantic.py +28 -0
  53. codegraph/security/__init__.py +3 -0
  54. codegraph/security/paths.py +35 -0
  55. codegraph/target_resolver.py +348 -0
  56. codegraph/task.py +637 -0
  57. codegraph_engine-2.1.1.dist-info/METADATA +334 -0
  58. codegraph_engine-2.1.1.dist-info/RECORD +62 -0
  59. codegraph_engine-2.1.1.dist-info/WHEEL +5 -0
  60. codegraph_engine-2.1.1.dist-info/entry_points.txt +2 -0
  61. codegraph_engine-2.1.1.dist-info/licenses/LICENSE +21 -0
  62. codegraph_engine-2.1.1.dist-info/top_level.txt +1 -0
@@ -0,0 +1,943 @@
1
+ """Incremental SQLite indexer with atomic per-file transactions and reference resolution.
2
+
3
+ Architectural Invariants:
4
+ 1. Normalized SQLite storage with PRAGMA foreign_keys = ON and ON DELETE CASCADE.
5
+ 2. Explicit FTS5 synchronization on every chunk insertion and deletion.
6
+ 3. Parse-failure safety: if a file has syntax errors, its last-known-good index state is
7
+ retained, marked 'parse_failed', and reported in freshness without corrupting the DB.
8
+ 4. Reference resolution runs globally after file parsing, turning raw source facts into
9
+ verified references, graph edges, and evidence.
10
+ 5. Parser version and schema version participate in cache and index validity.
11
+ """
12
+ from __future__ import annotations
13
+
14
+ import hashlib
15
+ import sqlite3
16
+ import time
17
+ from collections.abc import Iterator
18
+ from contextlib import contextmanager
19
+ from pathlib import Path
20
+
21
+ from codegraph.config import Settings
22
+ from codegraph.frameworks import RouteDetection
23
+ from codegraph.resolver import ReferenceResolver
24
+ from codegraph.resources import (
25
+ ResourceGovernor,
26
+ TaskPriority,
27
+ get_global_governor,
28
+ get_parse_cache,
29
+ )
30
+
31
+ from .models import (
32
+ CallRef,
33
+ ImportRef,
34
+ InheritanceRef,
35
+ Symbol,
36
+ )
37
+ from .parser import PARSER_VERSION, parse
38
+ from .scanner import scan
39
+
40
+ SCHEMA_VERSION = 6
41
+
42
+ SCHEMA = """
43
+ CREATE TABLE IF NOT EXISTS metadata (
44
+ key TEXT PRIMARY KEY,
45
+ value TEXT NOT NULL
46
+ );
47
+
48
+ CREATE TABLE IF NOT EXISTS codegraph_meta (
49
+ key TEXT PRIMARY KEY,
50
+ value TEXT NOT NULL
51
+ );
52
+
53
+ CREATE TABLE IF NOT EXISTS files (
54
+ path TEXT PRIMARY KEY,
55
+ hash TEXT NOT NULL,
56
+ language TEXT NOT NULL,
57
+ indexed_at INTEGER NOT NULL DEFAULT 0,
58
+ status TEXT NOT NULL DEFAULT 'ok',
59
+ parse_error TEXT,
60
+ last_valid_hash TEXT,
61
+ category TEXT NOT NULL DEFAULT 'SOURCE'
62
+ );
63
+
64
+ CREATE TABLE IF NOT EXISTS chunks (
65
+ id INTEGER PRIMARY KEY,
66
+ path TEXT NOT NULL,
67
+ language TEXT NOT NULL,
68
+ symbol TEXT,
69
+ symbol_type TEXT,
70
+ start_line INTEGER NOT NULL,
71
+ end_line INTEGER NOT NULL,
72
+ content TEXT NOT NULL,
73
+ hash TEXT NOT NULL,
74
+ FOREIGN KEY(path) REFERENCES files(path) ON DELETE CASCADE
75
+ );
76
+
77
+ CREATE VIRTUAL TABLE IF NOT EXISTS chunks_fts USING fts5(
78
+ content, path UNINDEXED, symbol UNINDEXED
79
+ );
80
+
81
+ CREATE TABLE IF NOT EXISTS symbols (
82
+ name TEXT NOT NULL,
83
+ qualified_name TEXT NOT NULL,
84
+ kind TEXT NOT NULL,
85
+ path TEXT NOT NULL,
86
+ start_line INTEGER NOT NULL,
87
+ end_line INTEGER NOT NULL,
88
+ decorators TEXT NOT NULL DEFAULT '',
89
+ id TEXT NOT NULL DEFAULT '',
90
+ canonical_id TEXT NOT NULL DEFAULT '',
91
+ language TEXT NOT NULL DEFAULT 'python',
92
+ module TEXT NOT NULL DEFAULT '',
93
+ scope TEXT NOT NULL DEFAULT '',
94
+ signature TEXT NOT NULL DEFAULT '',
95
+ content_hash TEXT NOT NULL DEFAULT '',
96
+ parent_symbol_id TEXT,
97
+ visibility TEXT NOT NULL DEFAULT 'public',
98
+ return_type TEXT,
99
+ parameter_count INTEGER,
100
+ documentation TEXT,
101
+ FOREIGN KEY(path) REFERENCES files(path) ON DELETE CASCADE
102
+ );
103
+
104
+ CREATE INDEX IF NOT EXISTS idx_symbols_name ON symbols (name);
105
+ CREATE INDEX IF NOT EXISTS idx_symbols_qualified ON symbols (qualified_name);
106
+ CREATE INDEX IF NOT EXISTS idx_symbols_canonical ON symbols (canonical_id);
107
+ CREATE INDEX IF NOT EXISTS idx_symbols_path ON symbols (path);
108
+ CREATE INDEX IF NOT EXISTS idx_symbols_parent ON symbols (parent_symbol_id);
109
+ CREATE INDEX IF NOT EXISTS idx_chunks_path ON chunks (path);
110
+ CREATE INDEX IF NOT EXISTS idx_chunks_symbol ON chunks (symbol);
111
+ CREATE INDEX IF NOT EXISTS idx_chunks_path_symbol ON chunks (path, symbol);
112
+
113
+ CREATE TABLE IF NOT EXISTS imports (
114
+ source_path TEXT NOT NULL,
115
+ module TEXT NOT NULL,
116
+ name TEXT,
117
+ alias TEXT,
118
+ full_name TEXT,
119
+ line INTEGER NOT NULL DEFAULT 1,
120
+ source_module TEXT NOT NULL DEFAULT '',
121
+ imported_module TEXT NOT NULL DEFAULT '',
122
+ imported_name TEXT,
123
+ local_name TEXT NOT NULL DEFAULT '',
124
+ import_type TEXT NOT NULL DEFAULT 'module',
125
+ resolved_path TEXT,
126
+ resolved_module TEXT,
127
+ FOREIGN KEY(source_path) REFERENCES files(path) ON DELETE CASCADE
128
+ );
129
+
130
+ CREATE INDEX IF NOT EXISTS idx_imports_source ON imports (source_path);
131
+ CREATE INDEX IF NOT EXISTS idx_imports_module ON imports (module);
132
+ CREATE INDEX IF NOT EXISTS idx_imports_imported_module ON imports (imported_module);
133
+
134
+ CREATE TABLE IF NOT EXISTS calls (
135
+ source_path TEXT NOT NULL,
136
+ callee TEXT NOT NULL,
137
+ qualified_callee TEXT,
138
+ line INTEGER NOT NULL DEFAULT 1,
139
+ confidence TEXT NOT NULL DEFAULT 'LOW',
140
+ source_symbol_id TEXT,
141
+ resolved_symbol_id TEXT,
142
+ FOREIGN KEY(source_path) REFERENCES files(path) ON DELETE CASCADE
143
+ );
144
+
145
+ CREATE INDEX IF NOT EXISTS idx_calls_source ON calls (source_path);
146
+ CREATE INDEX IF NOT EXISTS idx_calls_callee ON calls (callee);
147
+ CREATE INDEX IF NOT EXISTS idx_calls_resolved ON calls (resolved_symbol_id);
148
+ CREATE INDEX IF NOT EXISTS idx_calls_callee_source ON calls (callee, source_path);
149
+
150
+ CREATE TABLE IF NOT EXISTS inheritance (
151
+ source_symbol TEXT NOT NULL,
152
+ base_name TEXT NOT NULL,
153
+ relationship TEXT NOT NULL,
154
+ source_file TEXT NOT NULL,
155
+ line INTEGER NOT NULL DEFAULT 1,
156
+ source_canonical_id TEXT NOT NULL DEFAULT '',
157
+ FOREIGN KEY(source_file) REFERENCES files(path) ON DELETE CASCADE
158
+ );
159
+
160
+ CREATE INDEX IF NOT EXISTS idx_inh_source ON inheritance (source_file);
161
+
162
+ CREATE TABLE IF NOT EXISTS framework_routes (
163
+ endpoint_id TEXT PRIMARY KEY,
164
+ framework TEXT NOT NULL,
165
+ http_method TEXT NOT NULL,
166
+ route_path TEXT NOT NULL,
167
+ normalized_route TEXT NOT NULL,
168
+ handler_name TEXT NOT NULL,
169
+ handler_canonical_id TEXT NOT NULL,
170
+ file_path TEXT NOT NULL,
171
+ line INTEGER NOT NULL,
172
+ evidence TEXT NOT NULL,
173
+ confidence TEXT NOT NULL DEFAULT 'HIGH',
174
+ FOREIGN KEY(file_path) REFERENCES files(path) ON DELETE CASCADE
175
+ );
176
+
177
+ CREATE INDEX IF NOT EXISTS idx_routes_file ON framework_routes (file_path);
178
+ CREATE INDEX IF NOT EXISTS idx_routes_handler ON framework_routes (handler_canonical_id);
179
+ CREATE INDEX IF NOT EXISTS idx_routes_path ON framework_routes (route_path);
180
+
181
+ CREATE TABLE IF NOT EXISTS "references" (
182
+ id INTEGER PRIMARY KEY AUTOINCREMENT,
183
+ source_symbol_id TEXT NOT NULL,
184
+ target_symbol_id TEXT,
185
+ relationship TEXT NOT NULL,
186
+ confidence TEXT NOT NULL,
187
+ path TEXT NOT NULL,
188
+ start_line INTEGER NOT NULL,
189
+ end_line INTEGER NOT NULL,
190
+ evidence TEXT NOT NULL DEFAULT '',
191
+ source_hash TEXT NOT NULL DEFAULT '',
192
+ indexed_commit TEXT,
193
+ evidence_status TEXT NOT NULL DEFAULT 'current',
194
+ FOREIGN KEY(path) REFERENCES files(path) ON DELETE CASCADE
195
+ );
196
+
197
+ CREATE INDEX IF NOT EXISTS idx_refs_source ON "references" (source_symbol_id);
198
+ CREATE INDEX IF NOT EXISTS idx_refs_target ON "references" (target_symbol_id);
199
+ CREATE INDEX IF NOT EXISTS idx_refs_path ON "references" (path);
200
+ CREATE INDEX IF NOT EXISTS idx_refs_rel ON "references" (relationship);
201
+
202
+ CREATE TABLE IF NOT EXISTS graph_edges (
203
+ source TEXT NOT NULL,
204
+ target TEXT NOT NULL,
205
+ relationship TEXT NOT NULL,
206
+ confidence TEXT NOT NULL,
207
+ file TEXT NOT NULL,
208
+ start_line INTEGER NOT NULL,
209
+ end_line INTEGER NOT NULL,
210
+ evidence TEXT NOT NULL DEFAULT '',
211
+ evidence_id TEXT NOT NULL DEFAULT '',
212
+ source_hash TEXT NOT NULL DEFAULT '',
213
+ indexed_commit TEXT,
214
+ FOREIGN KEY(file) REFERENCES files(path) ON DELETE CASCADE
215
+ );
216
+
217
+ CREATE INDEX IF NOT EXISTS idx_edges_source ON graph_edges (source);
218
+ CREATE INDEX IF NOT EXISTS idx_edges_target ON graph_edges (target);
219
+ CREATE INDEX IF NOT EXISTS idx_edges_rel ON graph_edges (relationship);
220
+ CREATE INDEX IF NOT EXISTS idx_edges_file ON graph_edges (file);
221
+ CREATE INDEX IF NOT EXISTS idx_edges_evidence ON graph_edges (evidence_id);
222
+ CREATE INDEX IF NOT EXISTS idx_edges_target_rel ON graph_edges (target, relationship);
223
+ CREATE INDEX IF NOT EXISTS idx_edges_source_rel ON graph_edges (source, relationship);
224
+
225
+ CREATE TABLE IF NOT EXISTS context_cache (
226
+ cache_key TEXT PRIMARY KEY,
227
+ data TEXT NOT NULL,
228
+ created_at INTEGER NOT NULL
229
+ );
230
+ """
231
+
232
+
233
+ class Indexer:
234
+ def __init__(
235
+ self,
236
+ repository: Path,
237
+ settings: Settings | None = None,
238
+ governor: ResourceGovernor | None = None,
239
+ ) -> None:
240
+ self.repository = repository.resolve(strict=True)
241
+ self.settings = settings or Settings()
242
+ self.db_path = self.settings.db_path or self.repository / ".codegraph.sqlite3"
243
+ self.governor = governor or get_global_governor()
244
+
245
+ def connect(self) -> sqlite3.Connection:
246
+ con = sqlite3.connect(self.db_path)
247
+ con.row_factory = sqlite3.Row
248
+ con.execute("PRAGMA journal_mode = WAL")
249
+ con.execute("PRAGMA foreign_keys = ON")
250
+ con.execute("PRAGMA busy_timeout = 5000")
251
+ self._ensure_schema(con)
252
+ return con
253
+
254
+ @contextmanager
255
+ def session(self) -> Iterator[sqlite3.Connection]:
256
+ """Commit on success, roll back on failure, always close handle."""
257
+ con = self.connect()
258
+ try:
259
+ yield con
260
+ con.commit()
261
+ except BaseException:
262
+ con.rollback()
263
+ raise
264
+ finally:
265
+ con.close()
266
+
267
+ def _ensure_schema(self, con: sqlite3.Connection) -> None:
268
+ """Migrate schema via PRAGMA user_version non-destructively."""
269
+ current_version = con.execute("PRAGMA user_version").fetchone()[0]
270
+
271
+ if current_version == 0:
272
+ try:
273
+ row = con.execute("SELECT version FROM schema_version").fetchone()
274
+ if row:
275
+ current_version = int(row[0])
276
+ except sqlite3.OperationalError:
277
+ pass
278
+
279
+ target_version = SCHEMA_VERSION
280
+
281
+ if current_version < target_version:
282
+ if current_version > 0:
283
+ self._migrate_schema(con, current_version, target_version)
284
+ else:
285
+ con.executescript(SCHEMA)
286
+
287
+ con.execute(f"PRAGMA user_version = {target_version}")
288
+ try:
289
+ con.execute("DROP TABLE IF EXISTS schema_version")
290
+ except sqlite3.OperationalError:
291
+ pass
292
+ con.commit()
293
+
294
+ def _migrate_schema(
295
+ self, con: sqlite3.Connection, from_version: int, to_version: int
296
+ ) -> None:
297
+ """Non-destructive schema migration using ALTER TABLE."""
298
+ del from_version, to_version
299
+ # Ensure new tables and indexes are created
300
+ con.executescript(SCHEMA)
301
+
302
+ # Add missing columns to existing tables
303
+ def existing_columns(tbl: str) -> set[str]:
304
+ try:
305
+ return {
306
+ str(r["name"])
307
+ for r in con.execute(f"PRAGMA table_info({tbl})").fetchall()
308
+ }
309
+ except sqlite3.OperationalError:
310
+ return set()
311
+
312
+ files_cols = existing_columns("files")
313
+ if "status" not in files_cols:
314
+ con.execute("ALTER TABLE files ADD COLUMN status TEXT NOT NULL DEFAULT 'ok'")
315
+ if "parse_error" not in files_cols:
316
+ con.execute("ALTER TABLE files ADD COLUMN parse_error TEXT")
317
+ if "last_valid_hash" not in files_cols:
318
+ con.execute("ALTER TABLE files ADD COLUMN last_valid_hash TEXT")
319
+ if "category" not in files_cols:
320
+ con.execute("ALTER TABLE files ADD COLUMN category TEXT NOT NULL DEFAULT 'SOURCE'")
321
+
322
+ symbols_cols = existing_columns("symbols")
323
+ for col_def in (
324
+ ("id", "TEXT NOT NULL DEFAULT ''"),
325
+ ("canonical_id", "TEXT NOT NULL DEFAULT ''"),
326
+ ("language", "TEXT NOT NULL DEFAULT 'python'"),
327
+ ("module", "TEXT NOT NULL DEFAULT ''"),
328
+ ("scope", "TEXT NOT NULL DEFAULT ''"),
329
+ ("signature", "TEXT NOT NULL DEFAULT ''"),
330
+ ("content_hash", "TEXT NOT NULL DEFAULT ''"),
331
+ ("parent_symbol_id", "TEXT"),
332
+ ("visibility", "TEXT NOT NULL DEFAULT 'public'"),
333
+ ("return_type", "TEXT"),
334
+ ("parameter_count", "INTEGER"),
335
+ ("documentation", "TEXT"),
336
+ ):
337
+ if col_def[0] not in symbols_cols:
338
+ con.execute(f"ALTER TABLE symbols ADD COLUMN {col_def[0]} {col_def[1]}")
339
+
340
+ imports_cols = existing_columns("imports")
341
+ for col_def in (
342
+ ("source_module", "TEXT NOT NULL DEFAULT ''"),
343
+ ("imported_module", "TEXT NOT NULL DEFAULT ''"),
344
+ ("imported_name", "TEXT"),
345
+ ("local_name", "TEXT NOT NULL DEFAULT ''"),
346
+ ("import_type", "TEXT NOT NULL DEFAULT 'module'"),
347
+ ("resolved_path", "TEXT"),
348
+ ("resolved_module", "TEXT"),
349
+ ):
350
+ if col_def[0] not in imports_cols:
351
+ con.execute(f"ALTER TABLE imports ADD COLUMN {col_def[0]} {col_def[1]}")
352
+
353
+ files_cols = existing_columns("files")
354
+ if "category" not in files_cols:
355
+ con.execute("ALTER TABLE files ADD COLUMN category TEXT NOT NULL DEFAULT 'SOURCE'")
356
+
357
+ calls_cols = existing_columns("calls")
358
+ for col_def in (
359
+ ("source_symbol_id", "TEXT"),
360
+ ("resolved_symbol_id", "TEXT"),
361
+ ):
362
+ if col_def[0] not in calls_cols:
363
+ con.execute(f"ALTER TABLE calls ADD COLUMN {col_def[0]} {col_def[1]}")
364
+
365
+ def index(self) -> dict[str, int]:
366
+ """Perform an incremental indexing pass, followed by global reference resolution."""
367
+ indexed = unchanged = parse_failed_count = 0
368
+ from codegraph.freshness import current_commit, save_commit
369
+
370
+ with self.governor.task_scope(TaskPriority.BACKGROUND):
371
+ self.governor.set_indexing_active(True)
372
+ try:
373
+ files = scan(self.repository, self.settings.max_file_size, self.settings.exclude)
374
+ seen = {item.relative_path.as_posix() for item in files}
375
+ batch_limit = self.governor.policy.max_files_per_incremental_batch
376
+
377
+ with self.session() as con:
378
+ # Check parser version invalidation
379
+ last_parser_ver = None
380
+ try:
381
+ row = con.execute("SELECT value FROM metadata WHERE key='parser_version'").fetchone()
382
+ if row:
383
+ last_parser_ver = str(row[0])
384
+ except sqlite3.OperationalError:
385
+ pass
386
+
387
+ force_reparse = (last_parser_ver != PARSER_VERSION)
388
+
389
+ for idx, item in enumerate(files):
390
+ if idx > 0 and idx % batch_limit == 0:
391
+ self.governor.yield_if_needed(TaskPriority.BACKGROUND)
392
+
393
+ relative = item.relative_path.as_posix()
394
+ content = item.path.read_text(encoding="utf-8", errors="replace")
395
+ digest = hashlib.sha256(content.encode()).hexdigest()
396
+ old = con.execute(
397
+ "SELECT hash, status FROM files WHERE path=?", (relative,)
398
+ ).fetchone()
399
+
400
+ if old and old["hash"] == digest and not force_reparse:
401
+ if old["status"] == "parse_failed":
402
+ parse_failed_count += 1
403
+ else:
404
+ unchanged += 1
405
+ continue
406
+
407
+ cat_obj = getattr(item, "category", None)
408
+ cat_str = str(cat_obj.value) if cat_obj is not None and hasattr(cat_obj, "value") else "SOURCE"
409
+ success = self._replace_file(con, relative, item.language, content, digest, category=cat_str)
410
+ if success:
411
+ indexed += 1
412
+ else:
413
+ parse_failed_count += 1
414
+
415
+ stale = [
416
+ r[0]
417
+ for r in con.execute("SELECT path FROM files")
418
+ if r[0] not in seen
419
+ ]
420
+ for path in stale:
421
+ self._delete_file(con, path)
422
+
423
+ # Record metadata
424
+ head_commit = current_commit(self.repository)
425
+ now_ts = int(time.time())
426
+ save_commit(con, head_commit, now_ts)
427
+
428
+ # Increment index generation
429
+ current_gen = 0
430
+ try:
431
+ grow = con.execute("SELECT value FROM metadata WHERE key='index_generation'").fetchone()
432
+ if grow and grow[0]:
433
+ current_gen = int(grow[0])
434
+ except (sqlite3.OperationalError, ValueError):
435
+ pass
436
+ next_gen = current_gen + 1
437
+
438
+ for key, val in [
439
+ ("repository", str(self.repository)),
440
+ ("parser_version", PARSER_VERSION),
441
+ ("schema_version", str(SCHEMA_VERSION)),
442
+ ("index_generation", str(next_gen)),
443
+ ("index_timestamp", str(now_ts)),
444
+ ]:
445
+ con.execute(
446
+ "INSERT INTO metadata(key, value) VALUES(?,?) "
447
+ "ON CONFLICT(key) DO UPDATE SET value=excluded.value",
448
+ (key, val),
449
+ )
450
+
451
+ # Run global reference resolution pass if any files were changed or removed
452
+ if indexed > 0 or len(stale) > 0 or force_reparse:
453
+ self._run_global_resolution(con, head_commit)
454
+ # Invalidate context cache on index update
455
+ try:
456
+ con.execute("DELETE FROM context_cache")
457
+ except sqlite3.OperationalError:
458
+ pass
459
+ finally:
460
+ self.governor.set_indexing_active(False)
461
+
462
+ result: dict[str, int] = {
463
+ "scanned": len(files),
464
+ "indexed": indexed,
465
+ "unchanged": unchanged,
466
+ "removed": len(stale),
467
+ }
468
+ if parse_failed_count > 0:
469
+ result["parse_failed"] = parse_failed_count
470
+ return result
471
+
472
+ def _delete_file(self, con: sqlite3.Connection, path: str) -> None:
473
+ """Atomically delete all indexed facts for a file, maintaining FTS synchronization."""
474
+ ids = [r[0] for r in con.execute("SELECT id FROM chunks WHERE path=?", (path,))]
475
+ for chunk_id in ids:
476
+ con.execute("DELETE FROM chunks_fts WHERE rowid=?", (chunk_id,))
477
+ con.execute("DELETE FROM chunks WHERE path=?", (path,))
478
+ con.execute("DELETE FROM symbols WHERE path=?", (path,))
479
+ con.execute("DELETE FROM imports WHERE source_path=?", (path,))
480
+ con.execute("DELETE FROM calls WHERE source_path=?", (path,))
481
+ con.execute("DELETE FROM inheritance WHERE source_file=?", (path,))
482
+ con.execute("DELETE FROM framework_routes WHERE file_path=?", (path,))
483
+ con.execute("DELETE FROM 'references' WHERE path=?", (path,))
484
+ con.execute("DELETE FROM graph_edges WHERE file=?", (path,))
485
+ con.execute("DELETE FROM files WHERE path=?", (path,))
486
+
487
+ def _replace_file(
488
+ self,
489
+ con: sqlite3.Connection,
490
+ path: str,
491
+ language: str,
492
+ content: str,
493
+ digest: str,
494
+ category: str = "SOURCE",
495
+ ) -> bool:
496
+ """Parse and insert a file. If parse fails, preserves last-known-good state."""
497
+ parse_cache = get_parse_cache()
498
+ cached = parse_cache.get((digest, PARSER_VERSION, language))
499
+ if cached is not None:
500
+ result = cached
501
+ else:
502
+ result = parse(content, language, path)
503
+ parse_cache.put((digest, PARSER_VERSION, language), result)
504
+ now = int(time.time())
505
+
506
+ if result.parse_failed:
507
+ # Parse failure safety: check if file previously existed
508
+ old = con.execute("SELECT hash, status FROM files WHERE path=?", (path,)).fetchone()
509
+ if old:
510
+ # Retain last-known-good index records, mark file as parse_failed
511
+ con.execute(
512
+ "UPDATE files SET hash=?, status='parse_failed', parse_error=?, indexed_at=?, category=? WHERE path=?",
513
+ (digest, result.parse_error, now, category, path),
514
+ )
515
+ else:
516
+ con.execute(
517
+ "INSERT INTO files(path, hash, language, indexed_at, status, parse_error, category) VALUES (?, ?, ?, ?, ?, ?, ?)",
518
+ (path, digest, language, now, "parse_failed", result.parse_error, category),
519
+ )
520
+ return False
521
+
522
+ # Successful parse: clean up old file state
523
+ self._delete_file(con, path)
524
+ con.execute(
525
+ "INSERT INTO files(path, hash, language, indexed_at, status, parse_error, last_valid_hash, category) "
526
+ "VALUES (?, ?, ?, ?, 'ok', NULL, ?, ?)",
527
+ (path, digest, language, now, digest, category),
528
+ )
529
+
530
+ if category == "GENERATED":
531
+ # Suppress generated bundle noise: do not extract AST symbols or populate chunks_fts
532
+ return True
533
+
534
+ lines = content.splitlines()
535
+
536
+ if result.symbols:
537
+ for s in result.symbols:
538
+ dec_str = ",".join(s.decorators) if s.decorators else ""
539
+ con.execute(
540
+ "INSERT INTO symbols("
541
+ "name, qualified_name, kind, path, start_line, end_line, decorators, "
542
+ "id, canonical_id, language, module, scope, signature, content_hash, "
543
+ "parent_symbol_id, visibility, return_type, parameter_count, documentation) "
544
+ "VALUES (?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?,?)",
545
+ (
546
+ s.name,
547
+ s.qualified_name,
548
+ s.kind,
549
+ path,
550
+ s.start_line,
551
+ s.end_line,
552
+ dec_str,
553
+ s.id,
554
+ s.canonical_id,
555
+ s.language,
556
+ s.module,
557
+ s.scope,
558
+ s.signature,
559
+ s.content_hash,
560
+ s.parent_symbol_id,
561
+ s.visibility,
562
+ s.return_type,
563
+ s.parameter_count,
564
+ s.documentation,
565
+ ),
566
+ )
567
+ body = "\n".join(lines[s.start_line - 1 : s.end_line])
568
+ cur = con.execute(
569
+ "INSERT INTO chunks(path, language, symbol, symbol_type, start_line, end_line, content, hash) "
570
+ "VALUES (?,?,?,?,?,?,?,?)",
571
+ (
572
+ path,
573
+ language,
574
+ s.qualified_name,
575
+ s.kind,
576
+ s.start_line,
577
+ s.end_line,
578
+ body,
579
+ digest,
580
+ ),
581
+ )
582
+ con.execute(
583
+ "INSERT INTO chunks_fts(rowid, content, path, symbol) VALUES (?,?,?,?)",
584
+ (cur.lastrowid, body, path, s.qualified_name),
585
+ )
586
+ else:
587
+ cur = con.execute(
588
+ "INSERT INTO chunks(path, language, symbol, symbol_type, start_line, end_line, content, hash) "
589
+ "VALUES (?,?,?,?,?,?,?,?)",
590
+ (path, language, None, "module", 1, max(1, len(lines)), content, digest),
591
+ )
592
+ con.execute(
593
+ "INSERT INTO chunks_fts(rowid, content, path, symbol) VALUES (?,?,?,?)",
594
+ (cur.lastrowid, content, path, None),
595
+ )
596
+
597
+ if result.imports:
598
+ con.executemany(
599
+ "INSERT INTO imports(source_path, module, name, alias, full_name, line, "
600
+ "source_module, imported_module, imported_name, local_name, import_type) "
601
+ "VALUES (?,?,?,?,?,?,?,?,?,?,?)",
602
+ [
603
+ (
604
+ path,
605
+ imp.module,
606
+ imp.name,
607
+ imp.alias,
608
+ imp.full,
609
+ imp.line,
610
+ imp.source_module,
611
+ imp.imported_module,
612
+ imp.imported_name,
613
+ imp.local_name,
614
+ imp.import_type,
615
+ )
616
+ for imp in result.imports
617
+ ],
618
+ )
619
+
620
+ if result.calls:
621
+ con.executemany(
622
+ "INSERT INTO calls(source_path, callee, qualified_callee, line, confidence, source_symbol_id) "
623
+ "VALUES (?,?,?,?,?,?)",
624
+ [
625
+ (
626
+ path,
627
+ c.callee,
628
+ c.qualified_callee,
629
+ c.line,
630
+ c.confidence,
631
+ c.caller_canonical_id,
632
+ )
633
+ for c in result.calls
634
+ ],
635
+ )
636
+
637
+ if result.inheritance:
638
+ con.executemany(
639
+ "INSERT INTO inheritance(source_symbol, base_name, relationship, source_file, line, source_canonical_id) "
640
+ "VALUES (?,?,?,?,?,?)",
641
+ [
642
+ (
643
+ inh.source_symbol,
644
+ inh.base_name,
645
+ inh.relationship,
646
+ path,
647
+ inh.line,
648
+ inh.source_canonical_id,
649
+ )
650
+ for inh in result.inheritance
651
+ ],
652
+ )
653
+
654
+ if result.routes:
655
+ con.executemany(
656
+ "INSERT INTO framework_routes(endpoint_id, framework, http_method, route_path, "
657
+ "normalized_route, handler_name, handler_canonical_id, file_path, line, evidence, confidence) "
658
+ "VALUES (?,?,?,?,?,?,?,?,?,?,?)",
659
+ [
660
+ (
661
+ r.endpoint_id,
662
+ r.framework,
663
+ r.http_method,
664
+ r.route_path,
665
+ r.normalized_route,
666
+ r.handler_name,
667
+ r.handler_canonical_id,
668
+ path,
669
+ r.line,
670
+ r.evidence,
671
+ r.confidence,
672
+ )
673
+ for r in result.routes
674
+ ],
675
+ )
676
+
677
+ return True
678
+
679
+ def _run_global_resolution(
680
+ self, con: sqlite3.Connection, indexed_commit: str | None
681
+ ) -> None:
682
+ """Run repository-wide reference resolution, updating 'references' and 'graph_edges'."""
683
+ # 1. Load all files
684
+ file_rows = con.execute("SELECT path, hash FROM files WHERE status='ok'").fetchall()
685
+ known_files = {str(r["path"]) for r in file_rows}
686
+ file_hashes = {str(r["path"]): str(r["hash"]) for r in file_rows}
687
+
688
+ # 2. Load symbols
689
+ sym_rows = con.execute(
690
+ "SELECT name, qualified_name, kind, path, start_line, end_line, decorators, "
691
+ "id, canonical_id, language, module, scope, signature, content_hash, "
692
+ "parent_symbol_id, visibility, return_type, parameter_count, documentation "
693
+ "FROM symbols"
694
+ ).fetchall()
695
+ symbols = [
696
+ Symbol(
697
+ id=str(r["id"]),
698
+ canonical_id=str(r["canonical_id"]),
699
+ name=str(r["name"]),
700
+ qualified_name=str(r["qualified_name"]),
701
+ kind=str(r["kind"]),
702
+ start_line=int(r["start_line"]),
703
+ end_line=int(r["end_line"]),
704
+ file_path=str(r["path"]),
705
+ decorators=[d for d in str(r["decorators"]).split(",") if d],
706
+ language=str(r["language"]),
707
+ module=str(r["module"]),
708
+ path=str(r["path"]),
709
+ scope=str(r["scope"]),
710
+ signature=str(r["signature"]),
711
+ content_hash=str(r["content_hash"]),
712
+ parent_symbol_id=r["parent_symbol_id"],
713
+ visibility=str(r["visibility"]),
714
+ return_type=r["return_type"],
715
+ parameter_count=r["parameter_count"],
716
+ documentation=r["documentation"],
717
+ )
718
+ for r in sym_rows
719
+ ]
720
+
721
+ # 3. Load imports
722
+ imp_rows = con.execute(
723
+ "SELECT source_path, module, name, alias, full_name, line, "
724
+ "source_module, imported_module, imported_name, local_name, import_type "
725
+ "FROM imports"
726
+ ).fetchall()
727
+ imports = [
728
+ ImportRef(
729
+ module=str(r["module"]),
730
+ source_file=str(r["source_path"]),
731
+ name=r["name"],
732
+ alias=r["alias"],
733
+ full=r["full_name"],
734
+ line=int(r["line"]),
735
+ source_module=str(r["source_module"]),
736
+ imported_module=str(r["imported_module"]),
737
+ imported_name=r["imported_name"],
738
+ local_name=str(r["local_name"]),
739
+ import_type=str(r["import_type"]),
740
+ is_reexport=(str(r["import_type"]) == "reexport"),
741
+ )
742
+ for r in imp_rows
743
+ ]
744
+
745
+ # 4. Load calls
746
+ call_rows = con.execute(
747
+ "SELECT source_path, callee, qualified_callee, line, confidence, source_symbol_id "
748
+ "FROM calls"
749
+ ).fetchall()
750
+ calls = [
751
+ CallRef(
752
+ callee=str(r["callee"]),
753
+ source_file=str(r["source_path"]),
754
+ confidence=str(r["confidence"]),
755
+ qualified_callee=r["qualified_callee"],
756
+ line=int(r["line"]),
757
+ caller_canonical_id=r["source_symbol_id"],
758
+ )
759
+ for r in call_rows
760
+ ]
761
+
762
+ # 5. Load inheritance
763
+ inh_rows = con.execute(
764
+ "SELECT source_symbol, base_name, relationship, source_file, line, source_canonical_id "
765
+ "FROM inheritance"
766
+ ).fetchall()
767
+ inheritance = [
768
+ InheritanceRef(
769
+ source_symbol=str(r["source_symbol"]),
770
+ base_name=str(r["base_name"]),
771
+ relationship=str(r["relationship"]),
772
+ source_file=str(r["source_file"]),
773
+ line=int(r["line"]),
774
+ source_canonical_id=str(r["source_canonical_id"]),
775
+ )
776
+ for r in inh_rows
777
+ ]
778
+
779
+ # 6. Load framework routes
780
+ rt_rows = con.execute(
781
+ "SELECT endpoint_id, framework, http_method, route_path, normalized_route, "
782
+ "handler_name, handler_canonical_id, file_path, line, evidence, confidence "
783
+ "FROM framework_routes"
784
+ ).fetchall()
785
+ routes = [
786
+ RouteDetection(
787
+ framework=str(r["framework"]),
788
+ http_method=str(r["http_method"]),
789
+ route_path=str(r["route_path"]),
790
+ normalized_route=str(r["normalized_route"]),
791
+ handler_name=str(r["handler_name"]),
792
+ handler_canonical_id=str(r["handler_canonical_id"]),
793
+ file_path=str(r["file_path"]),
794
+ line=int(r["line"]),
795
+ evidence=str(r["evidence"]),
796
+ confidence=str(r["confidence"]),
797
+ )
798
+ for r in rt_rows
799
+ ]
800
+
801
+ # Run resolution engine
802
+ resolver = ReferenceResolver(
803
+ symbols=symbols,
804
+ imports=imports,
805
+ calls=calls,
806
+ inheritance=inheritance,
807
+ routes=routes,
808
+ known_files=known_files,
809
+ file_hashes=file_hashes,
810
+ indexed_commit=indexed_commit,
811
+ )
812
+ output = resolver.resolve_all()
813
+
814
+ # Update references table
815
+ con.execute("DELETE FROM 'references'")
816
+ con.executemany(
817
+ "INSERT INTO 'references'("
818
+ "source_symbol_id, target_symbol_id, relationship, confidence, path, "
819
+ "start_line, end_line, evidence, source_hash, indexed_commit, evidence_status) "
820
+ "VALUES (?,?,?,?,?,?,?,?,?,?,?)",
821
+ [
822
+ (
823
+ ref.source_symbol_id,
824
+ ref.target_symbol_id,
825
+ ref.relationship,
826
+ ref.confidence,
827
+ ref.path,
828
+ ref.start_line,
829
+ ref.end_line,
830
+ ref.evidence,
831
+ ref.source_hash,
832
+ ref.indexed_commit,
833
+ ref.evidence_status,
834
+ )
835
+ for ref in output.references
836
+ ],
837
+ )
838
+
839
+ # Update graph_edges table
840
+ con.execute("DELETE FROM graph_edges")
841
+ con.executemany(
842
+ "INSERT INTO graph_edges("
843
+ "source, target, relationship, confidence, file, start_line, end_line, "
844
+ "evidence, evidence_id, source_hash, indexed_commit) "
845
+ "VALUES (?,?,?,?,?,?,?,?,?,?,?)",
846
+ [
847
+ (
848
+ e.source,
849
+ e.target,
850
+ e.relationship,
851
+ e.confidence,
852
+ e.file,
853
+ e.start_line,
854
+ e.end_line,
855
+ e.evidence,
856
+ f"edge_{idx}",
857
+ file_hashes.get(e.file, ""),
858
+ indexed_commit,
859
+ )
860
+ for idx, e in enumerate(output.edges)
861
+ ],
862
+ )
863
+
864
+ # Update resolved fields on imports and calls tables
865
+ for (src_path, ln, loc_name), (tgt_path, tgt_mod) in output.resolved_imports.items():
866
+ con.execute(
867
+ "UPDATE imports SET resolved_path=?, resolved_module=? "
868
+ "WHERE source_path=? AND line=? AND (local_name=? OR alias=? OR name=?)",
869
+ (tgt_path, tgt_mod, src_path, ln, loc_name, loc_name, loc_name),
870
+ )
871
+
872
+ for (src_path, ln, callee_nm), (resolved_id, conf) in output.resolved_calls.items():
873
+ con.execute(
874
+ "UPDATE calls SET resolved_symbol_id=?, confidence=? "
875
+ "WHERE source_path=? AND line=? AND callee=?",
876
+ (resolved_id, conf, src_path, ln, callee_nm),
877
+ )
878
+
879
+
880
+ def check_database_health(con: sqlite3.Connection, repository: Path | None = None) -> dict[str, object]:
881
+ """Inspect database integrity, FTS synchronization, orphan records, and consistency."""
882
+ issues: list[str] = []
883
+
884
+ # 1. Foreign keys
885
+ fk_enabled = con.execute("PRAGMA foreign_keys").fetchone()[0]
886
+ if not fk_enabled:
887
+ issues.append("Foreign keys are disabled in SQLite connection.")
888
+
889
+ # 2. Schema and user version
890
+ user_ver = con.execute("PRAGMA user_version").fetchone()[0]
891
+ if user_ver < SCHEMA_VERSION:
892
+ issues.append(f"Schema version mismatch: database is v{user_ver}, expected v{SCHEMA_VERSION}.")
893
+
894
+ # 3. FTS synchronization check
895
+ chunks_count = con.execute("SELECT count(*) FROM chunks").fetchone()[0]
896
+ fts_count = con.execute("SELECT count(*) FROM chunks_fts").fetchone()[0]
897
+ if chunks_count != fts_count:
898
+ issues.append(f"FTS desynchronization: chunks table has {chunks_count} rows, chunks_fts has {fts_count} rows.")
899
+
900
+ # 4. Orphan symbols
901
+ orphan_syms = con.execute(
902
+ "SELECT count(*) FROM symbols s LEFT JOIN files f ON s.path = f.path WHERE f.path IS NULL"
903
+ ).fetchone()[0]
904
+ if orphan_syms > 0:
905
+ issues.append(f"Found {orphan_syms} orphan symbol(s) referencing non-existent files.")
906
+
907
+ # 5. Orphan references
908
+ orphan_refs = con.execute(
909
+ "SELECT count(*) FROM 'references' r LEFT JOIN files f ON r.path = f.path WHERE f.path IS NULL"
910
+ ).fetchone()[0]
911
+ if orphan_refs > 0:
912
+ issues.append(f"Found {orphan_refs} orphan reference(s) referencing non-existent files.")
913
+
914
+ # 6. Duplicate canonical IDs within same file
915
+ dups = con.execute(
916
+ "SELECT canonical_id, path, count(*) FROM symbols "
917
+ "GROUP BY canonical_id, path HAVING count(*) > 1"
918
+ ).fetchall()
919
+ if dups:
920
+ issues.append(f"Found {len(dups)} duplicate canonical symbol ID(s) in same file.")
921
+
922
+ # 7. Check parser version
923
+ parser_ver = None
924
+ try:
925
+ prow = con.execute("SELECT value FROM metadata WHERE key='parser_version'").fetchone()
926
+ if prow:
927
+ parser_ver = str(prow[0])
928
+ except sqlite3.OperationalError:
929
+ pass
930
+ if parser_ver and parser_ver != PARSER_VERSION:
931
+ issues.append(f"Parser version mismatch: indexed with {parser_ver}, active parser is {PARSER_VERSION}.")
932
+
933
+ status = "OK" if not issues else "ISSUES_FOUND"
934
+ return {
935
+ "status": status,
936
+ "database_version": user_ver,
937
+ "expected_version": SCHEMA_VERSION,
938
+ "parser_version": parser_ver or PARSER_VERSION,
939
+ "chunks_count": chunks_count,
940
+ "fts_count": fts_count,
941
+ "foreign_keys": bool(fk_enabled),
942
+ "issues": issues,
943
+ }