orbitkb 1.2.1__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (70) hide show
  1. orbitkb/__init__.py +1 -0
  2. orbitkb/cli.py +306 -0
  3. orbitkb/cli_progress.py +64 -0
  4. orbitkb/config.py +22 -0
  5. orbitkb/db/__init__.py +0 -0
  6. orbitkb/db/connection.py +29 -0
  7. orbitkb/db/repositories/__init__.py +3 -0
  8. orbitkb/db/repositories/_util.py +8 -0
  9. orbitkb/db/repositories/apis.py +81 -0
  10. orbitkb/db/repositories/architecture.py +43 -0
  11. orbitkb/db/repositories/change_surface.py +71 -0
  12. orbitkb/db/repositories/components.py +50 -0
  13. orbitkb/db/repositories/index_runs.py +35 -0
  14. orbitkb/db/repositories/indexed_files.py +34 -0
  15. orbitkb/db/repositories/messages.py +78 -0
  16. orbitkb/db/repositories/persistence.py +35 -0
  17. orbitkb/db/repositories/repositories.py +36 -0
  18. orbitkb/db/repositories/search.py +98 -0
  19. orbitkb/db/repositories/service_calls.py +212 -0
  20. orbitkb/db/repositories/services.py +64 -0
  21. orbitkb/db/repositories/verification.py +46 -0
  22. orbitkb/db/schema.sql +245 -0
  23. orbitkb/discovery/__init__.py +0 -0
  24. orbitkb/discovery/base.py +97 -0
  25. orbitkb/discovery/go_stack.py +150 -0
  26. orbitkb/discovery/hashing.py +45 -0
  27. orbitkb/discovery/integration_heuristics.py +78 -0
  28. orbitkb/discovery/jvm_stack.py +207 -0
  29. orbitkb/discovery/node_ts.py +181 -0
  30. orbitkb/discovery/python_stack.py +160 -0
  31. orbitkb/discovery/registry.py +23 -0
  32. orbitkb/discovery/scan_helpers.py +227 -0
  33. orbitkb/discovery/walker.py +72 -0
  34. orbitkb/export/__init__.py +0 -0
  35. orbitkb/export/markdown.py +101 -0
  36. orbitkb/export/mermaid.py +137 -0
  37. orbitkb/generation/__init__.py +0 -0
  38. orbitkb/generation/architecture.py +195 -0
  39. orbitkb/generation/backend_base.py +21 -0
  40. orbitkb/generation/change_surface.py +383 -0
  41. orbitkb/generation/claude_backend.py +61 -0
  42. orbitkb/generation/codex_backend.py +69 -0
  43. orbitkb/generation/freshness.py +23 -0
  44. orbitkb/generation/llm_harness.py +60 -0
  45. orbitkb/generation/next_queries.py +69 -0
  46. orbitkb/generation/orchestrator.py +491 -0
  47. orbitkb/generation/prompts/api_detail.md +37 -0
  48. orbitkb/generation/prompts/change_surface.md +21 -0
  49. orbitkb/generation/prompts/component.md +14 -0
  50. orbitkb/generation/prompts/messaging.md +26 -0
  51. orbitkb/generation/prompts/persistence.md +25 -0
  52. orbitkb/generation/prompts/service_overview.md +22 -0
  53. orbitkb/generation/provenance.py +9 -0
  54. orbitkb/generation/retrieval.py +104 -0
  55. orbitkb/generation/schemas/api_detail.schema.json +94 -0
  56. orbitkb/generation/schemas/change_surface.schema.json +27 -0
  57. orbitkb/generation/schemas/component.schema.json +11 -0
  58. orbitkb/generation/schemas/messaging.schema.json +37 -0
  59. orbitkb/generation/schemas/persistence.schema.json +36 -0
  60. orbitkb/generation/schemas/service_overview.schema.json +15 -0
  61. orbitkb/generation/verification.py +96 -0
  62. orbitkb/mcp/__init__.py +0 -0
  63. orbitkb/mcp/queries.py +321 -0
  64. orbitkb/mcp/server.py +229 -0
  65. orbitkb-1.2.1.dist-info/METADATA +968 -0
  66. orbitkb-1.2.1.dist-info/RECORD +70 -0
  67. orbitkb-1.2.1.dist-info/WHEEL +5 -0
  68. orbitkb-1.2.1.dist-info/entry_points.txt +2 -0
  69. orbitkb-1.2.1.dist-info/licenses/LICENSE +216 -0
  70. orbitkb-1.2.1.dist-info/top_level.txt +1 -0
orbitkb/__init__.py ADDED
@@ -0,0 +1 @@
1
+ __version__ = "1.2.1"
orbitkb/cli.py ADDED
@@ -0,0 +1,306 @@
1
+ from __future__ import annotations
2
+
3
+ import argparse
4
+ import json
5
+ import sys
6
+ from pathlib import Path
7
+
8
+ import orbitkb
9
+ from orbitkb.cli_progress import RichProgressReporter
10
+ from orbitkb.config import resolve_backend
11
+ from orbitkb.db.connection import DEFAULT_DB_PATH, open_db
12
+ from orbitkb.db.repositories import index_runs as index_runs_repo
13
+ from orbitkb.db.repositories import indexed_files as indexed_files_repo
14
+ from orbitkb.db.repositories import services as services_repo
15
+ from orbitkb.db.repositories import verification as verification_repo
16
+ from orbitkb.export.markdown import export_markdown
17
+ from orbitkb.export.mermaid import export_mermaid
18
+ from orbitkb.generation import change_surface
19
+ from orbitkb.generation.backend_base import GenerationError
20
+ from orbitkb.generation.orchestrator import DiscoveryError, index_path, index_service
21
+ from orbitkb.generation.verification import verify_change_surface
22
+
23
+
24
+ def _cmd_index(args: argparse.Namespace) -> int:
25
+ conn = open_db(args.db)
26
+ backend = resolve_backend(args.backend, args.model, args.claude_bare, args.codex_api_key)
27
+ try:
28
+ with RichProgressReporter() as progress:
29
+ results = index_path(
30
+ conn, Path(args.path), backend, service_override=args.service, force=args.force,
31
+ progress=progress, repository_name=args.repository_name,
32
+ )
33
+ except (DiscoveryError, GenerationError) as exc:
34
+ print(f"error: {exc}", file=sys.stderr)
35
+ return 1
36
+ for r in results:
37
+ print(f"{r.service_name}: status={r.status} files_changed={r.files_changed} llm_calls={r.llm_calls}")
38
+ return 0 if all(r.status == "ok" for r in results) else 1
39
+
40
+
41
+ def _cmd_update(args: argparse.Namespace) -> int:
42
+ conn = open_db(args.db)
43
+ row = services_repo.get_service_by_name(conn, args.service)
44
+ if row is None:
45
+ print(f"error: unknown service {args.service!r} (run `orbitkb list`)", file=sys.stderr)
46
+ return 1
47
+ root = Path(row["root_path"])
48
+ if not root.is_dir():
49
+ print(
50
+ f"error: root path for {args.service!r} no longer exists: {root}\n"
51
+ f" re-run `orbitkb index <newpath> --service {args.service}` instead.",
52
+ file=sys.stderr,
53
+ )
54
+ return 1
55
+ from orbitkb.discovery.registry import detector_for
56
+
57
+ detector = detector_for(root)
58
+ if detector is None:
59
+ print(f"error: {root} no longer matches any known stack", file=sys.stderr)
60
+ return 1
61
+ backend = resolve_backend(args.backend, args.model, args.claude_bare, args.codex_api_key)
62
+ with RichProgressReporter() as progress:
63
+ result = index_service(conn, args.service, root, detector, backend, force=args.force, progress=progress)
64
+ print(f"{result.service_name}: status={result.status} files_changed={result.files_changed} llm_calls={result.llm_calls}")
65
+ return 0 if result.status == "ok" else 1
66
+
67
+
68
+ def _cmd_list(args: argparse.Namespace) -> int:
69
+ conn = open_db(args.db)
70
+ rows = services_repo.list_services(conn)
71
+ if not rows:
72
+ print("(nenhum serviço indexado ainda)")
73
+ return 0
74
+ for r in rows:
75
+ print(f"{r['name']:<30} stack={r['stack'] or '?':<14} apis={r['api_count']:<3} {r['short_desc'] or ''}")
76
+ return 0
77
+
78
+
79
+ def _cmd_status(args: argparse.Namespace) -> int:
80
+ conn = open_db(args.db)
81
+ if args.service:
82
+ row = services_repo.get_service_by_name(conn, args.service)
83
+ if row is None:
84
+ print(f"error: unknown service {args.service!r}", file=sys.stderr)
85
+ return 1
86
+ hashes = indexed_files_repo.get_indexed_file_hashes(conn, row["id"])
87
+ print(f"service: {row['name']} ({row['stack']}) — {row['root_path']}")
88
+ print(f"last_commit: {row['last_commit']}")
89
+ print(f"indexed files: {len(hashes)}")
90
+ for run in index_runs_repo.recent_index_runs(conn, row["id"], limit=5):
91
+ print(
92
+ f" run#{run['id']} {run['started_at']} status={run['status']} backend={run['backend']} "
93
+ f"files_changed={run['files_changed']} llm_calls={run['llm_calls']} notes={run['notes']}"
94
+ )
95
+ else:
96
+ services = services_repo.list_services(conn)
97
+ print(f"services indexed: {len(services)}")
98
+ for run in index_runs_repo.recent_index_runs(conn, limit=10):
99
+ svc = services_repo.get_service_by_id(conn, run["service_id"]) if run["service_id"] else None
100
+ name = svc["name"] if svc else "?"
101
+ print(
102
+ f" run#{run['id']} service={name} status={run['status']} backend={run['backend']} "
103
+ f"files_changed={run['files_changed']} llm_calls={run['llm_calls']}"
104
+ )
105
+ verifications = verification_repo.latest_verifications(conn, limit=5)
106
+ if verifications:
107
+ print("recent change surface verifications:")
108
+ for v in verifications:
109
+ print(
110
+ f" run#{v['run_id']} repository={v['repository']} since={v['since_commit']} "
111
+ f"precision={v['precision']} recall={v['recall']}"
112
+ )
113
+ return 0
114
+
115
+
116
+ def _cmd_export(args: argparse.Namespace) -> int:
117
+ conn = open_db(args.db)
118
+ if args.format == "mermaid":
119
+ written = export_mermaid(conn, Path(args.out), service_filter=args.service)
120
+ else:
121
+ written = export_markdown(conn, Path(args.out), service_filter=args.service)
122
+ print(f"wrote {len(written)} files under {args.out}")
123
+ return 0
124
+
125
+
126
+ def _cmd_analyze(args: argparse.Namespace) -> int:
127
+ conn = open_db(args.db)
128
+ backend = resolve_backend(args.backend, args.model, args.claude_bare, args.codex_api_key)
129
+ result = change_surface.analyze_change_surface(conn, args.task, backend, hint_services=args.hint_services)
130
+ print(json.dumps(result, indent=2))
131
+ return 0
132
+
133
+
134
+ def _cmd_verify(args: argparse.Namespace) -> int:
135
+ conn = open_db(args.db)
136
+ result = verify_change_surface(conn, args.run_id, args.repository, args.since, record_feedback=args.record_feedback)
137
+ if "error" in result:
138
+ print(f"error: {result['error']}", file=sys.stderr)
139
+ return 1
140
+ print(f"predicted: {result['predicted']}")
141
+ print(f"actual: {result['actual']}")
142
+ print(f"true_positives: {result['true_positives']}")
143
+ print(f"false_positives: {result['false_positives']}")
144
+ print(f"false_negatives: {result['false_negatives']}")
145
+ print(f"precision: {result['precision']}")
146
+ print(f"recall: {result['recall']}")
147
+ if args.record_feedback:
148
+ print("feedback recorded for true/false positives")
149
+ return 0
150
+
151
+
152
+ def _cmd_serve(args: argparse.Namespace) -> int:
153
+ from orbitkb.mcp.server import build_server
154
+
155
+ backend = resolve_backend(args.backend, args.model, args.claude_bare, args.codex_api_key)
156
+ server = build_server(args.db, backend=backend)
157
+ server.run()
158
+ return 0
159
+
160
+
161
+ _TOP_LEVEL_EPILOG = """\
162
+ The orbitkb workflow is index -> ask -> verify:
163
+
164
+ 1. index Point it at a repo (or several) so it builds a System Knowledge Model.
165
+ 2. ask Register it as an MCP server (`serve`) and have an agent call
166
+ find_change_surface with an engineering task/epic, before it opens
167
+ any file, to get the likely blast radius with evidence + confidence.
168
+ 3. verify Once the change ships, check whether the prediction was right
169
+ against the real git diff, closing the feedback loop.
170
+
171
+ Examples:
172
+ orbitkb index ~/code/my-monorepo --repository-name my-monorepo
173
+ orbitkb serve --backend claude
174
+ orbitkb verify 3 --repository my-monorepo --since a1b2c3d
175
+
176
+ Run `orbitkb <command> --help` for a runnable example of any single command.
177
+ """
178
+
179
+
180
+ def build_parser() -> argparse.ArgumentParser:
181
+ parser = argparse.ArgumentParser(
182
+ prog="orbitkb", epilog=_TOP_LEVEL_EPILOG, formatter_class=argparse.RawDescriptionHelpFormatter,
183
+ )
184
+ parser.add_argument("--version", action="version", version=f"orbitkb {orbitkb.__version__}")
185
+ sub = parser.add_subparsers(dest="command", required=True)
186
+
187
+ def add_backend_args(p: argparse.ArgumentParser) -> None:
188
+ p.add_argument("--backend", choices=["claude", "codex"], default=None, help="LLM backend to shell out to headless (default: whichever CLI is on PATH)")
189
+ p.add_argument("--model", default=None, help="Override the backend's default model")
190
+ p.add_argument("--claude-bare", action="store_true", help="Use ANTHROPIC_API_KEY billing instead of the Claude Code subscription session")
191
+ p.add_argument("--codex-api-key", action="store_true", help="Use CODEX_API_KEY billing instead of the ChatGPT subscription session")
192
+ p.add_argument("--db", type=Path, default=DEFAULT_DB_PATH, help=f"SQLite database path (default: {DEFAULT_DB_PATH})")
193
+
194
+ p_index = sub.add_parser(
195
+ "index", help="Index a monorepo root or a single service repo",
196
+ epilog=(
197
+ "examples:\n"
198
+ " orbitkb index ~/code/orders-service\n"
199
+ " orbitkb index ~/code/my-monorepo --repository-name my-monorepo\n"
200
+ " orbitkb index . --service custom-name --force\n"
201
+ ),
202
+ formatter_class=argparse.RawDescriptionHelpFormatter,
203
+ )
204
+ p_index.add_argument("path", help="A single service's root, or a monorepo root containing several")
205
+ p_index.add_argument("--service", default=None, help="Override the inferred service name (only valid for a single-service path)")
206
+ p_index.add_argument("--repository-name", default=None, help="Explicit repository name; avoids collisions when indexing several repos into one shared DB")
207
+ p_index.add_argument("--force", action="store_true", help="Regenerate everything, ignoring file-hash skip")
208
+ add_backend_args(p_index)
209
+ p_index.set_defaults(func=_cmd_index)
210
+
211
+ p_update = sub.add_parser(
212
+ "update", help="Re-index one already-known service by name",
213
+ epilog="example:\n orbitkb update orders-service\n",
214
+ formatter_class=argparse.RawDescriptionHelpFormatter,
215
+ )
216
+ p_update.add_argument("service", help="Exact name shown by `orbitkb list`")
217
+ p_update.add_argument("--force", action="store_true", help="Regenerate everything, ignoring file-hash skip")
218
+ add_backend_args(p_update)
219
+ p_update.set_defaults(func=_cmd_update)
220
+
221
+ p_list = sub.add_parser("list", help="List indexed services")
222
+ p_list.add_argument("--db", type=Path, default=DEFAULT_DB_PATH, help=f"SQLite database path (default: {DEFAULT_DB_PATH})")
223
+ p_list.set_defaults(func=_cmd_list)
224
+
225
+ p_status = sub.add_parser(
226
+ "status", help="Show indexing status/history, and recent change surface verifications",
227
+ epilog="examples:\n orbitkb status\n orbitkb status orders-service\n",
228
+ formatter_class=argparse.RawDescriptionHelpFormatter,
229
+ )
230
+ p_status.add_argument("service", nargs="?", default=None, help="Show one service's indexing history instead of the whole DB's")
231
+ p_status.add_argument("--db", type=Path, default=DEFAULT_DB_PATH, help=f"SQLite database path (default: {DEFAULT_DB_PATH})")
232
+ p_status.set_defaults(func=_cmd_status)
233
+
234
+ p_export = sub.add_parser(
235
+ "export", help="Export the database to Markdown or Mermaid diagrams",
236
+ epilog=(
237
+ "examples:\n"
238
+ " orbitkb export md --out docs/\n"
239
+ " orbitkb export mermaid --out docs/\n"
240
+ ),
241
+ formatter_class=argparse.RawDescriptionHelpFormatter,
242
+ )
243
+ p_export.add_argument("format", choices=["md", "mermaid"], help="md: human-readable docs; mermaid: topology.mmd + one er.mmd per service")
244
+ p_export.add_argument("--out", default="docs", help="Output directory (default: docs)")
245
+ p_export.add_argument("--service", default=None, help="Export only this service instead of every indexed one")
246
+ p_export.add_argument("--db", type=Path, default=DEFAULT_DB_PATH, help=f"SQLite database path (default: {DEFAULT_DB_PATH})")
247
+ p_export.set_defaults(func=_cmd_export)
248
+
249
+ p_analyze = sub.add_parser(
250
+ "analyze", help="Run find_change_surface for a task and print the result as JSON",
251
+ epilog=(
252
+ "examples:\n"
253
+ " orbitkb analyze \"Add support for Pix in checkout\"\n"
254
+ " orbitkb analyze \"Add support for Pix in checkout\" --backend claude --db verify/sample_project.db\n"
255
+ " orbitkb analyze \"xyz internal cleanup\" --hint-services notification-service\n"
256
+ ),
257
+ formatter_class=argparse.RawDescriptionHelpFormatter,
258
+ )
259
+ p_analyze.add_argument("task", help="Free-text engineering task/epic, e.g. \"Add support for Pix in checkout\"")
260
+ p_analyze.add_argument("--hint-services", nargs="+", default=None, help="Anchor the search on these services even without a keyword match")
261
+ add_backend_args(p_analyze)
262
+ p_analyze.set_defaults(func=_cmd_analyze)
263
+
264
+ p_verify = sub.add_parser(
265
+ "verify", help="Compare a past find_change_surface run against what a repository's commits actually changed",
266
+ epilog=(
267
+ "example:\n"
268
+ " orbitkb verify 3 --repository my-monorepo --since a1b2c3d --record-feedback\n"
269
+ ),
270
+ formatter_class=argparse.RawDescriptionHelpFormatter,
271
+ )
272
+ p_verify.add_argument("run_id", type=int, help="run_id from a prior find_change_surface/analyze response")
273
+ p_verify.add_argument("--repository", required=True, help="Repository name, as shown by `orbitkb list`/`--repository-name` at index time")
274
+ p_verify.add_argument("--since", required=True, help="Commit the run was made against; actual changes are `git diff --since..HEAD`")
275
+ p_verify.add_argument("--record-feedback", action="store_true", help="Auto-record confirmed/rejected feedback for the predicted services")
276
+ p_verify.add_argument("--db", type=Path, default=DEFAULT_DB_PATH, help=f"SQLite database path (default: {DEFAULT_DB_PATH})")
277
+ p_verify.set_defaults(func=_cmd_verify)
278
+
279
+ p_serve = sub.add_parser(
280
+ "serve", help="Run the MCP server (stdio)",
281
+ epilog=(
282
+ "example (register with an MCP client, e.g. Claude Code/Codex):\n"
283
+ " orbitkb serve --backend claude\n"
284
+ " orbitkb serve --db ~/.orbitkb/orbitkb.db --backend codex\n"
285
+ ),
286
+ formatter_class=argparse.RawDescriptionHelpFormatter,
287
+ )
288
+ p_serve.add_argument("--transport", choices=["stdio"], default="stdio", help="MCP transport (only stdio is supported today)")
289
+ add_backend_args(p_serve) # find_change_surface is the only tool that uses a backend
290
+ p_serve.set_defaults(func=_cmd_serve)
291
+
292
+ return parser
293
+
294
+
295
+ def main(argv: list[str] | None = None) -> int:
296
+ parser = build_parser()
297
+ args = parser.parse_args(argv)
298
+ try:
299
+ return args.func(args)
300
+ except KeyboardInterrupt:
301
+ print("\ncancelado pelo usuário (Ctrl+C)", file=sys.stderr)
302
+ return 130
303
+
304
+
305
+ if __name__ == "__main__":
306
+ raise SystemExit(main())
@@ -0,0 +1,64 @@
1
+ """Rich-based terminal progress for `orbitkb index`/`update`.
2
+
3
+ Kept separate from generation/orchestrator.py so the generation logic never depends
4
+ on a UI library — it only calls the small ProgressReporter protocol.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ from rich.progress import (
9
+ BarColumn,
10
+ MofNCompleteColumn,
11
+ Progress,
12
+ SpinnerColumn,
13
+ TaskProgressColumn,
14
+ TextColumn,
15
+ TimeElapsedColumn,
16
+ )
17
+
18
+ _STATUS_STYLE = {
19
+ "ok": "[green]ok[/green]",
20
+ "failed": "[bold red]falhou[/bold red]",
21
+ "skipped": "[dim]sem mudanças[/dim]",
22
+ }
23
+
24
+
25
+ class RichProgressReporter:
26
+ """One growing bar per service; finished services stay on screen at 100%, so a
27
+ multi-service run visibly fills up from top to bottom. TimeElapsedColumn keeps
28
+ ticking on the active row even while a single slow LLM call is in flight, so a
29
+ stalled step is visibly still "alive" (elapsed climbing) versus truly hung
30
+ (spinner frozen too, which only happens if the process itself died).
31
+ """
32
+
33
+ def __init__(self) -> None:
34
+ self._progress = Progress(
35
+ SpinnerColumn(),
36
+ TextColumn("[bold]{task.fields[service]}[/bold]"),
37
+ BarColumn(),
38
+ TaskProgressColumn(),
39
+ MofNCompleteColumn(),
40
+ TextColumn("{task.fields[detail]}"),
41
+ TimeElapsedColumn(),
42
+ )
43
+ self._tasks: dict[str, int] = {}
44
+
45
+ def __enter__(self) -> "RichProgressReporter":
46
+ self._progress.__enter__()
47
+ return self
48
+
49
+ def __exit__(self, *exc: object) -> None:
50
+ self._progress.__exit__(*exc)
51
+
52
+ def service_started(self, service: str, total_units: int) -> None:
53
+ task_id = self._progress.add_task(service, total=max(total_units, 1), service=service, detail="iniciando…")
54
+ self._tasks[service] = task_id
55
+
56
+ def unit_started(self, service: str, label: str) -> None:
57
+ self._progress.update(self._tasks[service], detail=f"gerando: {label}…")
58
+
59
+ def unit_finished(self, service: str, label: str, status: str) -> None:
60
+ style = _STATUS_STYLE.get(status, status)
61
+ self._progress.update(self._tasks[service], advance=1, detail=f"{label}: {style}")
62
+
63
+ def service_finished(self, service: str) -> None:
64
+ self._progress.update(self._tasks[service], detail="[bold green]concluído[/bold green]")
orbitkb/config.py ADDED
@@ -0,0 +1,22 @@
1
+ from __future__ import annotations
2
+
3
+ import os
4
+ from pathlib import Path
5
+
6
+ from orbitkb.generation.backend_base import LLMBackend
7
+ from orbitkb.generation.claude_backend import ClaudeBackend
8
+ from orbitkb.generation.codex_backend import CodexBackend
9
+
10
+ DEFAULT_BACKEND = "claude"
11
+ DEFAULT_DB_PATH = Path.home() / ".orbitkb" / "orbitkb.db"
12
+
13
+
14
+ def resolve_backend(
15
+ name: str | None, model: str | None = None, claude_bare: bool = False, codex_api_key: bool = False
16
+ ) -> LLMBackend:
17
+ chosen = name or os.environ.get("ORBITKB_BACKEND", DEFAULT_BACKEND)
18
+ if chosen == "claude":
19
+ return ClaudeBackend(model=model, bare=claude_bare)
20
+ if chosen == "codex":
21
+ return CodexBackend(model=model, api_key=codex_api_key)
22
+ raise ValueError(f"Unknown backend: {chosen!r} (expected 'claude' or 'codex')")
orbitkb/db/__init__.py ADDED
File without changes
@@ -0,0 +1,29 @@
1
+ from __future__ import annotations
2
+
3
+ import sqlite3
4
+ from importlib import resources
5
+ from pathlib import Path
6
+
7
+ SCHEMA_VERSION = "2"
8
+ DEFAULT_DB_PATH = Path.home() / ".orbitkb" / "orbitkb.db"
9
+
10
+
11
+ def open_db(db_path: Path | None = None) -> sqlite3.Connection:
12
+ path = db_path or DEFAULT_DB_PATH
13
+ path.parent.mkdir(parents=True, exist_ok=True)
14
+ conn = sqlite3.connect(str(path))
15
+ conn.row_factory = sqlite3.Row
16
+ conn.execute("PRAGMA foreign_keys = ON")
17
+ _init_schema(conn)
18
+ return conn
19
+
20
+
21
+ def _init_schema(conn: sqlite3.Connection) -> None:
22
+ schema_sql = resources.files("orbitkb.db").joinpath("schema.sql").read_text()
23
+ conn.executescript(schema_sql)
24
+ row = conn.execute("SELECT value FROM schema_meta WHERE key = 'schema_version'").fetchone()
25
+ if row is None:
26
+ conn.execute(
27
+ "INSERT INTO schema_meta (key, value) VALUES ('schema_version', ?)", (SCHEMA_VERSION,)
28
+ )
29
+ conn.commit()
@@ -0,0 +1,3 @@
1
+ """One module per aggregate, each owning the SQL for its own tables. These are the
2
+ only modules allowed to run SQL against the orbitkb database (replaces the single
3
+ db/repository.py, which had grown to own nine unrelated aggregates)."""
@@ -0,0 +1,8 @@
1
+ """Shared helper for every repository module."""
2
+ from __future__ import annotations
3
+
4
+ from datetime import datetime, timezone
5
+
6
+
7
+ def now() -> str:
8
+ return datetime.now(timezone.utc).isoformat()
@@ -0,0 +1,81 @@
1
+ """The `apis` and `api_validations` tables: one microservice's HTTP endpoints and
2
+ the validation/authorization rules attached to each."""
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import sqlite3
7
+
8
+ from ._util import now
9
+
10
+
11
+ def upsert_api(
12
+ conn: sqlite3.Connection,
13
+ service_id: int,
14
+ method: str,
15
+ path: str,
16
+ summary: str,
17
+ description: str,
18
+ response_shape: dict,
19
+ evidence: list[dict],
20
+ request_shape: list[dict] | None = None,
21
+ ) -> int:
22
+ row = conn.execute(
23
+ "SELECT id FROM apis WHERE service_id = ? AND method = ? AND path = ?",
24
+ (service_id, method, path),
25
+ ).fetchone()
26
+ payload = (
27
+ summary, description, json.dumps(response_shape), json.dumps(request_shape or []),
28
+ json.dumps(evidence), now(),
29
+ )
30
+ if row is not None:
31
+ conn.execute(
32
+ """UPDATE apis SET summary = ?, description = ?, response_shape = ?, request_shape = ?,
33
+ evidence_json = ?, updated_at = ? WHERE id = ?""",
34
+ (*payload, row["id"]),
35
+ )
36
+ api_id = row["id"]
37
+ else:
38
+ cur = conn.execute(
39
+ """INSERT INTO apis (service_id, method, path, summary, description, response_shape,
40
+ request_shape, evidence_json, updated_at) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?)""",
41
+ (service_id, method, path, *payload),
42
+ )
43
+ api_id = cur.lastrowid
44
+ conn.commit()
45
+ return api_id
46
+
47
+
48
+ def replace_api_validations(conn: sqlite3.Connection, api_id: int, validations: list[dict]) -> None:
49
+ conn.execute("DELETE FROM api_validations WHERE api_id = ?", (api_id,))
50
+ conn.executemany(
51
+ "INSERT INTO api_validations (api_id, kind, description) VALUES (?, ?, ?)",
52
+ [(api_id, v["kind"], v["description"]) for v in validations],
53
+ )
54
+ conn.commit()
55
+
56
+
57
+ def prune_apis_not_in(conn: sqlite3.Connection, service_id: int, keep_keys: set[tuple[str, str]]) -> None:
58
+ rows = conn.execute("SELECT id, method, path FROM apis WHERE service_id = ?", (service_id,)).fetchall()
59
+ for row in rows:
60
+ if (row["method"], row["path"]) not in keep_keys:
61
+ conn.execute("DELETE FROM apis WHERE id = ?", (row["id"],))
62
+ conn.commit()
63
+
64
+
65
+ def get_api_by_key(conn: sqlite3.Connection, service_id: int, method: str, path: str) -> sqlite3.Row | None:
66
+ return conn.execute(
67
+ "SELECT * FROM apis WHERE service_id = ? AND method = ? AND path = ?", (service_id, method, path)
68
+ ).fetchone()
69
+
70
+
71
+ def list_apis(conn: sqlite3.Connection, service_id: int) -> list[sqlite3.Row]:
72
+ return conn.execute(
73
+ "SELECT method, path, summary, description, evidence_json FROM apis WHERE service_id = ? ORDER BY path, method",
74
+ (service_id,),
75
+ ).fetchall()
76
+
77
+
78
+ def list_validations_for_api(conn: sqlite3.Connection, api_id: int) -> list[sqlite3.Row]:
79
+ return conn.execute(
80
+ "SELECT kind, description FROM api_validations WHERE api_id = ? ORDER BY kind", (api_id,)
81
+ ).fetchall()
@@ -0,0 +1,43 @@
1
+ """The `architecture_runs`/`architecture_findings` tables: deterministic, whole-graph
2
+ structural findings recomputed after every index/update. See generation/architecture.py
3
+ for the detectors themselves — this module only persists and reads their output."""
4
+ from __future__ import annotations
5
+
6
+ import json
7
+ import sqlite3
8
+
9
+ from ._util import now
10
+
11
+
12
+ def start_run(conn: sqlite3.Connection, services_indexed: int) -> int:
13
+ cur = conn.execute(
14
+ "INSERT INTO architecture_runs (created_at, services_indexed) VALUES (?, ?)",
15
+ (now(), services_indexed),
16
+ )
17
+ conn.commit()
18
+ return cur.lastrowid
19
+
20
+
21
+ def record_finding(
22
+ conn: sqlite3.Connection, run_id: int, kind: str, severity: str, services: list[str], reason: str,
23
+ detail: dict | None = None,
24
+ ) -> None:
25
+ conn.execute(
26
+ """INSERT INTO architecture_findings (run_id, kind, severity, services_json, detail_json, reason)
27
+ VALUES (?, ?, ?, ?, ?, ?)""",
28
+ (run_id, kind, severity, json.dumps(services), json.dumps(detail or {}), reason),
29
+ )
30
+ conn.commit()
31
+
32
+
33
+ def latest_run_id(conn: sqlite3.Connection) -> int | None:
34
+ row = conn.execute("SELECT id FROM architecture_runs ORDER BY id DESC LIMIT 1").fetchone()
35
+ return row["id"] if row else None
36
+
37
+
38
+ def list_findings(conn: sqlite3.Connection, run_id: int) -> list[sqlite3.Row]:
39
+ return conn.execute(
40
+ """SELECT kind, severity, services_json, detail_json, reason
41
+ FROM architecture_findings WHERE run_id = ? ORDER BY severity DESC, kind""",
42
+ (run_id,),
43
+ ).fetchall()
@@ -0,0 +1,71 @@
1
+ """The change-surface audit trail: `change_surface_runs`, `change_surface_findings`
2
+ and `change_surface_feedback` — every find_change_surface call and the outcome
3
+ feedback agents report back (see generation/change_surface.py, which is the
4
+ task-inference logic; this module is only the storage for it)."""
5
+ from __future__ import annotations
6
+
7
+ import json
8
+ import sqlite3
9
+
10
+ from ._util import now
11
+
12
+ _ROLE_BY_RESULT_KEY = {
13
+ "primary": "primary",
14
+ "secondary": "secondary",
15
+ "no_change_hint": "no_change",
16
+ "external_integrations": "external_integration",
17
+ "unmapped_internal_hint": "unmapped_internal",
18
+ }
19
+
20
+
21
+ def record_change_surface_run(conn: sqlite3.Connection, task_text: str, backend: str, result: dict) -> int:
22
+ cur = conn.execute(
23
+ "INSERT INTO change_surface_runs (task_text, backend, created_at) VALUES (?, ?, ?)",
24
+ (task_text, backend, now()),
25
+ )
26
+ run_id = cur.lastrowid
27
+ rows = [
28
+ (run_id, f["service"], role, f.get("reason"), f.get("confidence"), json.dumps(f.get("evidence", [])))
29
+ for key, role in _ROLE_BY_RESULT_KEY.items()
30
+ for f in result.get(key, [])
31
+ ]
32
+ conn.executemany(
33
+ """INSERT INTO change_surface_findings (run_id, service, role, reason, confidence, evidence_json)
34
+ VALUES (?, ?, ?, ?, ?, ?)""",
35
+ rows,
36
+ )
37
+ conn.commit()
38
+ return run_id
39
+
40
+
41
+ def get_change_surface_run(conn: sqlite3.Connection, run_id: int) -> sqlite3.Row | None:
42
+ return conn.execute("SELECT * FROM change_surface_runs WHERE id = ?", (run_id,)).fetchone()
43
+
44
+
45
+ def list_change_surface_runs(conn: sqlite3.Connection, limit: int = 10) -> list[sqlite3.Row]:
46
+ return conn.execute("SELECT * FROM change_surface_runs ORDER BY id DESC LIMIT ?", (limit,)).fetchall()
47
+
48
+
49
+ def list_change_surface_findings(conn: sqlite3.Connection, run_id: int) -> list[sqlite3.Row]:
50
+ return conn.execute(
51
+ "SELECT * FROM change_surface_findings WHERE run_id = ? ORDER BY id", (run_id,)
52
+ ).fetchall()
53
+
54
+
55
+ def record_change_surface_feedback(conn: sqlite3.Connection, run_id: int, service: str, outcome: str) -> None:
56
+ conn.execute(
57
+ "INSERT INTO change_surface_feedback (run_id, service, outcome, recorded_at) VALUES (?, ?, ?, ?)",
58
+ (run_id, service, outcome, now()),
59
+ )
60
+ conn.commit()
61
+
62
+
63
+ def get_feedback_stats(conn: sqlite3.Connection, service: str) -> dict[str, int]:
64
+ row = conn.execute(
65
+ """SELECT
66
+ SUM(CASE WHEN outcome = 'confirmed' THEN 1 ELSE 0 END) AS confirmed,
67
+ SUM(CASE WHEN outcome = 'rejected' THEN 1 ELSE 0 END) AS rejected
68
+ FROM change_surface_feedback WHERE service = ?""",
69
+ (service,),
70
+ ).fetchone()
71
+ return {"confirmed": row["confirmed"] or 0, "rejected": row["rejected"] or 0}