graphforge-neo4j 0.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- graphforge/__init__.py +12 -0
- graphforge/__main__.py +6 -0
- graphforge/cli.py +402 -0
- graphforge/core/__init__.py +23 -0
- graphforge/core/config.py +124 -0
- graphforge/core/cypher.py +155 -0
- graphforge/core/neo4j_writer.py +230 -0
- graphforge/db/__init__.py +66 -0
- graphforge/db/base.py +313 -0
- graphforge/db/ingest.py +199 -0
- graphforge/db/mssql.py +85 -0
- graphforge/db/mysql.py +80 -0
- graphforge/db/postgres.py +58 -0
- graphforge/db/vds.py +137 -0
- graphforge/git/__init__.py +5 -0
- graphforge/git/clone.py +112 -0
- graphforge/git/discover.py +48 -0
- graphforge/git/history.py +206 -0
- graphforge/git/ingest.py +442 -0
- graphforge/git/parsers/__init__.py +10 -0
- graphforge/git/parsers/generic.py +78 -0
- graphforge/git/parsers/golang.py +291 -0
- graphforge/git/parsers/java.py +148 -0
- graphforge/git/parsers/pom.py +54 -0
- graphforge/git/parsers/python.py +374 -0
- graphforge/git/parsers/typescript.py +387 -0
- graphforge/git/scan.py +180 -0
- graphforge/link/__init__.py +10 -0
- graphforge/link/passes.py +128 -0
- graphforge/mcp/__init__.py +5 -0
- graphforge/mcp/server.py +694 -0
- graphforge/py.typed +0 -0
- graphforge/schema/db_schema.cypher +24 -0
- graphforge/schema/git_schema.cypher +42 -0
- graphforge/schema/vds_schema.cypher +15 -0
- graphforge/ui/__init__.py +5 -0
- graphforge/ui/dashboard.html +731 -0
- graphforge/ui/server.py +412 -0
- graphforge_neo4j-0.2.0.dist-info/METADATA +541 -0
- graphforge_neo4j-0.2.0.dist-info/RECORD +44 -0
- graphforge_neo4j-0.2.0.dist-info/WHEEL +5 -0
- graphforge_neo4j-0.2.0.dist-info/entry_points.txt +2 -0
- graphforge_neo4j-0.2.0.dist-info/licenses/LICENSE +21 -0
- graphforge_neo4j-0.2.0.dist-info/top_level.txt +1 -0
graphforge/__init__.py
ADDED
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"""graphforge — forge Git repositories and relational databases into a Neo4j knowledge graph.
|
|
2
|
+
|
|
3
|
+
Published on PyPI as ``graphforge-neo4j``; imported as ``graphforge``.
|
|
4
|
+
"""
|
|
5
|
+
from __future__ import annotations
|
|
6
|
+
|
|
7
|
+
try:
|
|
8
|
+
from importlib.metadata import version
|
|
9
|
+
|
|
10
|
+
__version__ = version("graphforge-neo4j")
|
|
11
|
+
except ImportError: # not installed (source/editable/dev) — fall back to the literal
|
|
12
|
+
__version__ = "0.1.0"
|
graphforge/__main__.py
ADDED
graphforge/cli.py
ADDED
|
@@ -0,0 +1,402 @@
|
|
|
1
|
+
"""graphforge command-line interface.
|
|
2
|
+
|
|
3
|
+
graphforge init # create graph constraints & indexes
|
|
4
|
+
graphforge git [...] # ingest git repositories (structure + history)
|
|
5
|
+
graphforge db [...] # ingest relational database schemas
|
|
6
|
+
graphforge mcp # serve the graph over MCP
|
|
7
|
+
graphforge verify # connect and report node counts per label
|
|
8
|
+
"""
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import argparse
|
|
12
|
+
import json
|
|
13
|
+
import logging
|
|
14
|
+
import os
|
|
15
|
+
import sys
|
|
16
|
+
|
|
17
|
+
from . import __version__
|
|
18
|
+
from .core.config import Settings, load_settings
|
|
19
|
+
from .core.neo4j_writer import Neo4jWriter, load_schema
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _add_common(p: argparse.ArgumentParser) -> None:
|
|
23
|
+
p.add_argument("--env", metavar="FILE", help="path to a .env file to load")
|
|
24
|
+
p.add_argument("--neo4j-uri", dest="neo4j_uri")
|
|
25
|
+
p.add_argument("--neo4j-user", dest="neo4j_user")
|
|
26
|
+
p.add_argument("--neo4j-password", dest="neo4j_password")
|
|
27
|
+
p.add_argument("--neo4j-database", dest="neo4j_database")
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
def _add_writer_opts(p: argparse.ArgumentParser) -> None:
|
|
31
|
+
p.add_argument("--emit", metavar="FILE",
|
|
32
|
+
help="write a replayable .cypher script instead of pushing to Neo4j")
|
|
33
|
+
p.add_argument("--dry-run", action="store_true",
|
|
34
|
+
help="build operations but neither connect nor write (just count)")
|
|
35
|
+
p.add_argument("--no-schema", action="store_true",
|
|
36
|
+
help="skip creating constraints/indexes before ingesting")
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def _settings(args) -> Settings:
|
|
40
|
+
s = load_settings(getattr(args, "env", None))
|
|
41
|
+
if getattr(args, "neo4j_uri", None):
|
|
42
|
+
s.neo4j.uri = args.neo4j_uri
|
|
43
|
+
if getattr(args, "neo4j_user", None):
|
|
44
|
+
s.neo4j.user = args.neo4j_user
|
|
45
|
+
if getattr(args, "neo4j_password", None):
|
|
46
|
+
s.neo4j.password = args.neo4j_password
|
|
47
|
+
if getattr(args, "neo4j_database", None):
|
|
48
|
+
s.neo4j.database = args.neo4j_database
|
|
49
|
+
return s
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _writer(s: Settings, args) -> Neo4jWriter:
|
|
53
|
+
return Neo4jWriter(s.neo4j, emit_path=getattr(args, "emit", None),
|
|
54
|
+
dry_run=getattr(args, "dry_run", False))
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _report(writer: Neo4jWriter, extra: str = "") -> None:
|
|
58
|
+
where = {
|
|
59
|
+
"emit": f"wrote {writer.ops_written} operations -> {writer.emit_path}",
|
|
60
|
+
"dry-run": f"dry run: {writer.ops_written} operations (nothing written)",
|
|
61
|
+
"push": f"pushed {writer.ops_written} operations to Neo4j",
|
|
62
|
+
}[writer.mode]
|
|
63
|
+
print(f"[graphforge] {where}{(' | ' + extra) if extra else ''}")
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
# ----------------------------------------------------------------------------
|
|
67
|
+
# git
|
|
68
|
+
# ----------------------------------------------------------------------------
|
|
69
|
+
def _resolve_git_sources(args, git_settings) -> list[dict]:
|
|
70
|
+
specs: list[dict] = []
|
|
71
|
+
if args.config:
|
|
72
|
+
with open(args.config, encoding="utf-8") as fh:
|
|
73
|
+
data = json.load(fh)
|
|
74
|
+
default_branch = data.get("defaultBranch", git_settings.default_branch)
|
|
75
|
+
for repo in data.get("repositories", []):
|
|
76
|
+
repo.setdefault("branch", default_branch)
|
|
77
|
+
specs.append(repo)
|
|
78
|
+
for item in args.sources:
|
|
79
|
+
if "://" in item or item.endswith(".git"):
|
|
80
|
+
specs.append({"url": item, "name": args.name, "branch": args.branch})
|
|
81
|
+
else:
|
|
82
|
+
specs.append({"path": item, "name": args.name})
|
|
83
|
+
if not specs and git_settings.gitlab_server and git_settings.gitlab_group_id:
|
|
84
|
+
from .git.discover import gitlab_group_repos
|
|
85
|
+
specs = gitlab_group_repos(
|
|
86
|
+
git_settings.gitlab_server, git_settings.gitlab_group_id,
|
|
87
|
+
git_settings.gitlab_token, git_settings.default_branch,
|
|
88
|
+
since_days=getattr(args, "since", 0) or 0,
|
|
89
|
+
)
|
|
90
|
+
return specs
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
def cmd_git(args) -> int:
|
|
94
|
+
from .git.ingest import GitIngestor
|
|
95
|
+
|
|
96
|
+
s = _settings(args)
|
|
97
|
+
specs = _resolve_git_sources(args, s.git)
|
|
98
|
+
if not specs:
|
|
99
|
+
print("nothing to ingest: pass a path/URL, -c config.json, or set GITLAB_* env", file=sys.stderr)
|
|
100
|
+
return 2
|
|
101
|
+
with _writer(s, args) as w:
|
|
102
|
+
gi = GitIngestor(w, s.git)
|
|
103
|
+
if not args.no_schema:
|
|
104
|
+
gi.apply_schema()
|
|
105
|
+
stats = gi.ingest(
|
|
106
|
+
specs,
|
|
107
|
+
include_lines=args.lines or s.neo4j.include_lines,
|
|
108
|
+
with_structure=not args.no_structure,
|
|
109
|
+
with_history=not args.no_history,
|
|
110
|
+
replace=args.replace,
|
|
111
|
+
since_commit=getattr(args, "since_commit", "") or "",
|
|
112
|
+
)
|
|
113
|
+
extra = f"{stats['repos']} repos, {stats['files']} files, {stats['commits']} commits"
|
|
114
|
+
if stats.get("failed"):
|
|
115
|
+
extra += f", {stats['failed']} failed (see log with -v)"
|
|
116
|
+
_report(w, extra)
|
|
117
|
+
return 0
|
|
118
|
+
|
|
119
|
+
|
|
120
|
+
# ----------------------------------------------------------------------------
|
|
121
|
+
# db
|
|
122
|
+
# ----------------------------------------------------------------------------
|
|
123
|
+
def _default_schemas(engine: str, db_settings) -> list[str]:
|
|
124
|
+
"""PG_SCHEMAS is a PostgreSQL default; other engines discover their own."""
|
|
125
|
+
return list(db_settings.pg_schemas) if (engine or "").lower().startswith(("pg", "postgres")) else []
|
|
126
|
+
|
|
127
|
+
|
|
128
|
+
def _resolve_db_sources(args, db_settings) -> list[dict]:
|
|
129
|
+
sample_rows = getattr(args, "sample_rows", 0) or 0
|
|
130
|
+
if args.config:
|
|
131
|
+
with open(args.config, encoding="utf-8") as fh:
|
|
132
|
+
data = json.load(fh)
|
|
133
|
+
sources = data.get("sources", [])
|
|
134
|
+
for src in sources:
|
|
135
|
+
if "password" not in src and src.get("passwordEnv"):
|
|
136
|
+
src["password"] = os.getenv(src["passwordEnv"], "")
|
|
137
|
+
if sample_rows:
|
|
138
|
+
src["sampleRows"] = sample_rows
|
|
139
|
+
return sources
|
|
140
|
+
|
|
141
|
+
url = getattr(args, "url", None) or os.getenv("DB_URL")
|
|
142
|
+
if url:
|
|
143
|
+
from .db import parse_db_url
|
|
144
|
+
src = parse_db_url(url)
|
|
145
|
+
if args.databases:
|
|
146
|
+
src["databases"] = [d.strip() for d in args.databases.split(",") if d.strip()]
|
|
147
|
+
if args.driver:
|
|
148
|
+
src["driver"] = args.driver
|
|
149
|
+
if args.schemas:
|
|
150
|
+
src["schemas"] = [s.strip() for s in args.schemas.split(",") if s.strip()]
|
|
151
|
+
src["sampleRows"] = sample_rows
|
|
152
|
+
return [src]
|
|
153
|
+
|
|
154
|
+
engine = args.engine or db_settings.engine
|
|
155
|
+
databases = (args.databases.split(",") if args.databases else db_settings.names)
|
|
156
|
+
return [{
|
|
157
|
+
"engine": engine,
|
|
158
|
+
"host": args.host or db_settings.host,
|
|
159
|
+
"port": args.port or db_settings.port,
|
|
160
|
+
"user": args.user or db_settings.user,
|
|
161
|
+
"password": args.password or db_settings.password,
|
|
162
|
+
# empty -> auto-discover every non-system database on the server
|
|
163
|
+
"databases": [d.strip() for d in databases if d.strip()],
|
|
164
|
+
"driver": args.driver or db_settings.mssql_driver,
|
|
165
|
+
"schemas": (args.schemas.split(",") if args.schemas
|
|
166
|
+
else _default_schemas(engine, db_settings)),
|
|
167
|
+
"sampleRows": sample_rows,
|
|
168
|
+
}]
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
def cmd_db(args) -> int:
|
|
172
|
+
from .db import DbIngestor
|
|
173
|
+
|
|
174
|
+
have_conn = bool(getattr(args, "url", None) or args.config or args.host
|
|
175
|
+
or os.getenv("DB_URL") or os.getenv("DB_HOST"))
|
|
176
|
+
if not have_conn:
|
|
177
|
+
print("provide a connection: --url <db-url>, or --host …, or -c config.json "
|
|
178
|
+
"(or set DB_URL / DB_HOST)", file=sys.stderr)
|
|
179
|
+
return 2
|
|
180
|
+
s = _settings(args)
|
|
181
|
+
sources = _resolve_db_sources(args, s.db)
|
|
182
|
+
with _writer(s, args) as w:
|
|
183
|
+
di = DbIngestor(w)
|
|
184
|
+
if not args.no_schema:
|
|
185
|
+
di.apply_schema()
|
|
186
|
+
stats = di.ingest_sources(sources, replace=args.replace)
|
|
187
|
+
extra = f"{stats['databases']} databases, {stats['tables']} tables, {stats['columns']} columns"
|
|
188
|
+
if stats.get("failed"):
|
|
189
|
+
extra += f", {stats['failed']} failed (see log with -v)"
|
|
190
|
+
_report(w, extra)
|
|
191
|
+
return 0
|
|
192
|
+
|
|
193
|
+
|
|
194
|
+
# ----------------------------------------------------------------------------
|
|
195
|
+
# init / verify / mcp
|
|
196
|
+
# ----------------------------------------------------------------------------
|
|
197
|
+
def cmd_init(args) -> int:
|
|
198
|
+
s = _settings(args)
|
|
199
|
+
with _writer(s, args) as w:
|
|
200
|
+
n = w.apply_schema(load_schema("git_schema.cypher"))
|
|
201
|
+
n += w.apply_schema(load_schema("db_schema.cypher"))
|
|
202
|
+
print(f"[graphforge] applied {n} schema statements ({w.mode})")
|
|
203
|
+
return 0
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def cmd_verify(args) -> int:
|
|
207
|
+
s = _settings(args)
|
|
208
|
+
with Neo4jWriter(s.neo4j) as w:
|
|
209
|
+
counts = w.label_counts()
|
|
210
|
+
if not counts:
|
|
211
|
+
print("[graphforge] no nodes found (is the graph empty, or wrong database?)")
|
|
212
|
+
return 0
|
|
213
|
+
print("[graphforge] node counts by label:")
|
|
214
|
+
for label, count in sorted(counts.items(), key=lambda kv: -kv[1]):
|
|
215
|
+
print(f" {label:20s} {count}")
|
|
216
|
+
return 0
|
|
217
|
+
|
|
218
|
+
|
|
219
|
+
def cmd_link(args) -> int:
|
|
220
|
+
from .link import PASSES, LinkRunner
|
|
221
|
+
|
|
222
|
+
selected = [name for name in PASSES if getattr(args, name.replace("-", "_"))]
|
|
223
|
+
if not selected:
|
|
224
|
+
selected = list(PASSES) # default: run every pass
|
|
225
|
+
s = _settings(args)
|
|
226
|
+
with _writer(s, args) as w:
|
|
227
|
+
n = LinkRunner(w).run(selected, min_name_len=args.min_table_name_len)
|
|
228
|
+
_report(w, f"{n} link passes: {', '.join(selected)}")
|
|
229
|
+
return 0
|
|
230
|
+
|
|
231
|
+
|
|
232
|
+
def cmd_vds(args) -> int:
|
|
233
|
+
from .db.vds import VdsIngestor
|
|
234
|
+
|
|
235
|
+
s = _settings(args)
|
|
236
|
+
database = args.database or (s.db.names[0] if s.db.names else "")
|
|
237
|
+
if not database:
|
|
238
|
+
print("VDS import needs --database (or DB_NAMES)", file=sys.stderr)
|
|
239
|
+
return 2
|
|
240
|
+
with _writer(s, args) as w:
|
|
241
|
+
vi = VdsIngestor(w)
|
|
242
|
+
if not args.no_schema:
|
|
243
|
+
vi.apply_schema()
|
|
244
|
+
stats = vi.ingest(
|
|
245
|
+
engine=args.engine or s.db.engine, host=args.host or s.db.host,
|
|
246
|
+
port=args.port or s.db.port, user=args.user or s.db.user,
|
|
247
|
+
password=args.password or s.db.password, database=database,
|
|
248
|
+
driver=args.driver or s.db.mssql_driver,
|
|
249
|
+
)
|
|
250
|
+
_report(w, f"{stats['services']} VDS services, {stats['queries']} queries, "
|
|
251
|
+
f"{stats['whereFields']} where-fields")
|
|
252
|
+
return 0
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
def cmd_status(args) -> int:
|
|
256
|
+
s = _settings(args)
|
|
257
|
+
with Neo4jWriter(s.neo4j) as w:
|
|
258
|
+
rows = w.repository_status()
|
|
259
|
+
if not rows:
|
|
260
|
+
print("[graphforge] no repositories ingested yet")
|
|
261
|
+
return 0
|
|
262
|
+
print(f"{'repository':30s} {'status':12s} {'files':>7s} {'commits':>8s} lastIngestedAt")
|
|
263
|
+
for r in rows:
|
|
264
|
+
print(f"{(r.get('name') or ''):30s} {(r.get('status') or ''):12s} "
|
|
265
|
+
f"{(r.get('files') or 0):>7} {(r.get('commits') or 0):>8} {r.get('lastIngestedAt') or ''}")
|
|
266
|
+
return 0
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def cmd_ui(args) -> int:
|
|
270
|
+
from .ui import serve
|
|
271
|
+
|
|
272
|
+
s = _settings(args)
|
|
273
|
+
serve(s, host=args.host, port=args.port)
|
|
274
|
+
return 0
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
def cmd_mcp(args) -> int:
|
|
278
|
+
from .mcp.server import run
|
|
279
|
+
|
|
280
|
+
s = _settings(args)
|
|
281
|
+
run(s.neo4j)
|
|
282
|
+
return 0
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
# ----------------------------------------------------------------------------
|
|
286
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
287
|
+
from .link import passes as link_passes # for the --min-table-name-len default
|
|
288
|
+
|
|
289
|
+
parser = argparse.ArgumentParser(prog="graphforge", description=__doc__,
|
|
290
|
+
formatter_class=argparse.RawDescriptionHelpFormatter)
|
|
291
|
+
parser.add_argument("--version", action="version", version=f"graphforge {__version__}")
|
|
292
|
+
parser.add_argument("-v", "--verbose", action="store_true")
|
|
293
|
+
sub = parser.add_subparsers(dest="command", required=True)
|
|
294
|
+
|
|
295
|
+
p_init = sub.add_parser("init", help="create graph constraints and indexes")
|
|
296
|
+
_add_common(p_init)
|
|
297
|
+
_add_writer_opts(p_init)
|
|
298
|
+
p_init.set_defaults(func=cmd_init)
|
|
299
|
+
|
|
300
|
+
p_git = sub.add_parser("git", help="ingest git repositories (code structure + history)")
|
|
301
|
+
p_git.add_argument("sources", nargs="*", help="local repo paths and/or git URLs")
|
|
302
|
+
p_git.add_argument("-c", "--config", help="repositories.json")
|
|
303
|
+
p_git.add_argument("--name", help="name for a single repo passed positionally")
|
|
304
|
+
p_git.add_argument("--branch", help="branch for a single URL")
|
|
305
|
+
p_git.add_argument("--lines", action="store_true", help="store per-line :Line nodes (heavy)")
|
|
306
|
+
p_git.add_argument("--no-structure", action="store_true", help="skip code structure")
|
|
307
|
+
p_git.add_argument("--no-history", action="store_true", help="skip commit history")
|
|
308
|
+
p_git.add_argument("--replace", action="store_true", help="delete each repo's existing subgraph first")
|
|
309
|
+
p_git.add_argument("--since", type=int, default=0, metavar="DAYS",
|
|
310
|
+
help="GitLab discovery: only repos active in the last N days")
|
|
311
|
+
p_git.add_argument("--since-commit", dest="since_commit", metavar="SHA", default="",
|
|
312
|
+
help="incremental ingest: only commits after SHA (or 'auto' to "
|
|
313
|
+
"continue from :Repository.lastCommit)")
|
|
314
|
+
_add_common(p_git)
|
|
315
|
+
_add_writer_opts(p_git)
|
|
316
|
+
p_git.set_defaults(func=cmd_git)
|
|
317
|
+
|
|
318
|
+
p_db = sub.add_parser("db", help="ingest relational database schemas")
|
|
319
|
+
p_db.add_argument("-c", "--config", help="databases.json")
|
|
320
|
+
p_db.add_argument("--url", help="connection URL, e.g. postgresql://user:pass@host:5432/dbname "
|
|
321
|
+
"(omit the database to graph every non-system database)")
|
|
322
|
+
p_db.add_argument("--engine", help="mysql | postgres | mssql")
|
|
323
|
+
p_db.add_argument("--host")
|
|
324
|
+
p_db.add_argument("--port", type=int)
|
|
325
|
+
p_db.add_argument("--user")
|
|
326
|
+
p_db.add_argument("--password")
|
|
327
|
+
p_db.add_argument("--databases", help="comma-separated database names")
|
|
328
|
+
p_db.add_argument("--driver", help="MSSQL ODBC driver name")
|
|
329
|
+
p_db.add_argument("--schemas", help="comma-separated schema filter (postgres/mssql)")
|
|
330
|
+
p_db.add_argument("--sample-rows", dest="sample_rows", type=int, default=0, metavar="N",
|
|
331
|
+
help="opt-in profiling: COUNT(*) every table into :Table.approxRows, "
|
|
332
|
+
"and for tables of at most N rows also COUNT(DISTINCT col) into "
|
|
333
|
+
":Column.approxCardinality (default 0 = off)")
|
|
334
|
+
p_db.add_argument("--replace", action="store_true", help="delete each database's existing subgraph first")
|
|
335
|
+
_add_common(p_db)
|
|
336
|
+
_add_writer_opts(p_db)
|
|
337
|
+
p_db.set_defaults(func=cmd_db)
|
|
338
|
+
|
|
339
|
+
p_vds = sub.add_parser("vds", help="import a VDS (virtual data service) catalog")
|
|
340
|
+
p_vds.add_argument("--engine", help="mysql | postgres | mssql")
|
|
341
|
+
p_vds.add_argument("--host")
|
|
342
|
+
p_vds.add_argument("--port", type=int)
|
|
343
|
+
p_vds.add_argument("--user")
|
|
344
|
+
p_vds.add_argument("--password")
|
|
345
|
+
p_vds.add_argument("--database", help="database holding the VDS catalog tables")
|
|
346
|
+
p_vds.add_argument("--driver", help="MSSQL ODBC driver name")
|
|
347
|
+
_add_common(p_vds)
|
|
348
|
+
_add_writer_opts(p_vds)
|
|
349
|
+
p_vds.set_defaults(func=cmd_vds)
|
|
350
|
+
|
|
351
|
+
p_link = sub.add_parser("link", help="create code<->database links over the loaded graph")
|
|
352
|
+
p_link.add_argument("--maps-to", action="store_true", help="JPA entity -> Table")
|
|
353
|
+
p_link.add_argument("--based-on", action="store_true", help="View -> Table (same DB)")
|
|
354
|
+
p_link.add_argument("--uses-table", action="store_true", help="StoredProcedure -> Table (same DB)")
|
|
355
|
+
p_link.add_argument("--cross-db", action="store_true", help="View -> Table (different DB)")
|
|
356
|
+
p_link.add_argument("--min-table-name-len", dest="min_table_name_len", type=int,
|
|
357
|
+
default=link_passes.DEFAULT_MIN_NAME_LEN, metavar="N",
|
|
358
|
+
help="ignore table names shorter than N characters in the SQL-text "
|
|
359
|
+
f"passes (default {link_passes.DEFAULT_MIN_NAME_LEN})")
|
|
360
|
+
_add_common(p_link)
|
|
361
|
+
_add_writer_opts(p_link)
|
|
362
|
+
p_link.set_defaults(func=cmd_link)
|
|
363
|
+
|
|
364
|
+
p_verify = sub.add_parser("verify", help="report node counts per label")
|
|
365
|
+
_add_common(p_verify)
|
|
366
|
+
p_verify.set_defaults(func=cmd_verify)
|
|
367
|
+
|
|
368
|
+
p_status = sub.add_parser("status", help="show per-repository ingest status")
|
|
369
|
+
_add_common(p_status)
|
|
370
|
+
p_status.set_defaults(func=cmd_status)
|
|
371
|
+
|
|
372
|
+
p_ui = sub.add_parser("ui", help="serve a local web dashboard (config + load status)")
|
|
373
|
+
p_ui.add_argument("--host", default="127.0.0.1")
|
|
374
|
+
p_ui.add_argument("--port", type=int, default=8000)
|
|
375
|
+
_add_common(p_ui)
|
|
376
|
+
p_ui.set_defaults(func=cmd_ui)
|
|
377
|
+
|
|
378
|
+
p_mcp = sub.add_parser("mcp", help="serve the knowledge graph over MCP (stdio)")
|
|
379
|
+
_add_common(p_mcp)
|
|
380
|
+
p_mcp.set_defaults(func=cmd_mcp)
|
|
381
|
+
|
|
382
|
+
return parser
|
|
383
|
+
|
|
384
|
+
|
|
385
|
+
def main(argv: list[str] | None = None) -> int:
|
|
386
|
+
parser = build_parser()
|
|
387
|
+
args = parser.parse_args(argv)
|
|
388
|
+
logging.basicConfig(
|
|
389
|
+
level=logging.INFO if args.verbose else logging.WARNING,
|
|
390
|
+
format="%(levelname)s %(name)s: %(message)s",
|
|
391
|
+
)
|
|
392
|
+
try:
|
|
393
|
+
return args.func(args)
|
|
394
|
+
except (ValueError, RuntimeError, FileNotFoundError) as exc:
|
|
395
|
+
# Expected, user-facing failures (bad engine, missing driver, missing
|
|
396
|
+
# config file) exit cleanly instead of dumping a traceback.
|
|
397
|
+
print(f"error: {exc}", file=sys.stderr)
|
|
398
|
+
return 2
|
|
399
|
+
|
|
400
|
+
|
|
401
|
+
if __name__ == "__main__":
|
|
402
|
+
sys.exit(main())
|
|
@@ -0,0 +1,23 @@
|
|
|
1
|
+
"""Core: configuration, graph operation model, Cypher builders, and the Neo4j writer."""
|
|
2
|
+
|
|
3
|
+
from .cypher import (
|
|
4
|
+
NodeRef,
|
|
5
|
+
Operation,
|
|
6
|
+
escape_cypher_string,
|
|
7
|
+
lit,
|
|
8
|
+
merge_node,
|
|
9
|
+
merge_rel,
|
|
10
|
+
set_label,
|
|
11
|
+
)
|
|
12
|
+
from .neo4j_writer import Neo4jWriter
|
|
13
|
+
|
|
14
|
+
__all__ = [
|
|
15
|
+
"Neo4jWriter",
|
|
16
|
+
"NodeRef",
|
|
17
|
+
"Operation",
|
|
18
|
+
"escape_cypher_string",
|
|
19
|
+
"lit",
|
|
20
|
+
"merge_node",
|
|
21
|
+
"merge_rel",
|
|
22
|
+
"set_label",
|
|
23
|
+
]
|
|
@@ -0,0 +1,124 @@
|
|
|
1
|
+
"""Environment-driven configuration. Nothing here is hardcoded to any host.
|
|
2
|
+
|
|
3
|
+
Values are read from the process environment (optionally seeded from a local
|
|
4
|
+
`.env` file via python-dotenv). CLI flags may override individual fields.
|
|
5
|
+
"""
|
|
6
|
+
from __future__ import annotations
|
|
7
|
+
|
|
8
|
+
import os
|
|
9
|
+
from dataclasses import dataclass, field
|
|
10
|
+
|
|
11
|
+
try:
|
|
12
|
+
from dotenv import load_dotenv
|
|
13
|
+
except ImportError: # pragma: no cover - dotenv is a core dep but keep import soft
|
|
14
|
+
def load_dotenv(*_args, **_kwargs): # type: ignore
|
|
15
|
+
return False
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _bool(value: str | None, default: bool = False) -> bool:
|
|
19
|
+
if value is None:
|
|
20
|
+
return default
|
|
21
|
+
return value.strip().lower() in {"1", "true", "yes", "on"}
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _int(value: str | None, default: int) -> int:
|
|
25
|
+
try:
|
|
26
|
+
return int(str(value).strip())
|
|
27
|
+
except (TypeError, ValueError):
|
|
28
|
+
return default
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
def _csv(value: str | None) -> list[str]:
|
|
32
|
+
if not value:
|
|
33
|
+
return []
|
|
34
|
+
return [item.strip() for item in value.split(",") if item.strip()]
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@dataclass
|
|
38
|
+
class Neo4jSettings:
|
|
39
|
+
uri: str = "bolt://127.0.0.1:7687"
|
|
40
|
+
user: str = "neo4j"
|
|
41
|
+
password: str = ""
|
|
42
|
+
database: str = "neo4j"
|
|
43
|
+
batch_size: int = 500
|
|
44
|
+
include_lines: bool = False
|
|
45
|
+
|
|
46
|
+
@classmethod
|
|
47
|
+
def from_env(cls) -> Neo4jSettings:
|
|
48
|
+
return cls(
|
|
49
|
+
uri=os.getenv("NEO4J_URI", cls.uri),
|
|
50
|
+
user=os.getenv("NEO4J_USER", cls.user),
|
|
51
|
+
password=os.getenv("NEO4J_PASSWORD", cls.password),
|
|
52
|
+
database=os.getenv("NEO4J_DATABASE", cls.database),
|
|
53
|
+
batch_size=_int(os.getenv("GF_BATCH_SIZE"), cls.batch_size),
|
|
54
|
+
include_lines=_bool(os.getenv("GF_INCLUDE_LINES"), cls.include_lines),
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
@dataclass
|
|
59
|
+
class GitSettings:
|
|
60
|
+
repo_dir: str = "./repos"
|
|
61
|
+
username: str = ""
|
|
62
|
+
password: str = ""
|
|
63
|
+
token: str = ""
|
|
64
|
+
gitlab_server: str = ""
|
|
65
|
+
gitlab_group_id: str = ""
|
|
66
|
+
gitlab_token: str = ""
|
|
67
|
+
default_branch: str = "main"
|
|
68
|
+
history_limit: int = 0
|
|
69
|
+
|
|
70
|
+
@classmethod
|
|
71
|
+
def from_env(cls) -> GitSettings:
|
|
72
|
+
return cls(
|
|
73
|
+
repo_dir=os.getenv("GF_REPO_DIR", cls.repo_dir),
|
|
74
|
+
username=os.getenv("GIT_USERNAME", cls.username),
|
|
75
|
+
password=os.getenv("GIT_PASSWORD", cls.password),
|
|
76
|
+
token=os.getenv("GIT_TOKEN", cls.token),
|
|
77
|
+
gitlab_server=os.getenv("GITLAB_SERVER", cls.gitlab_server),
|
|
78
|
+
gitlab_group_id=os.getenv("GITLAB_GROUP_ID", cls.gitlab_group_id),
|
|
79
|
+
gitlab_token=os.getenv("GITLAB_TOKEN", cls.gitlab_token),
|
|
80
|
+
default_branch=os.getenv("GIT_DEFAULT_BRANCH", cls.default_branch),
|
|
81
|
+
history_limit=_int(os.getenv("GF_HISTORY_LIMIT"), cls.history_limit),
|
|
82
|
+
)
|
|
83
|
+
|
|
84
|
+
|
|
85
|
+
@dataclass
|
|
86
|
+
class DbSettings:
|
|
87
|
+
engine: str = "mysql"
|
|
88
|
+
host: str = "127.0.0.1"
|
|
89
|
+
port: int = 3306
|
|
90
|
+
user: str = ""
|
|
91
|
+
password: str = ""
|
|
92
|
+
names: list[str] = field(default_factory=list)
|
|
93
|
+
mssql_driver: str = "ODBC Driver 18 for SQL Server"
|
|
94
|
+
pg_schemas: list[str] = field(default_factory=lambda: ["public"])
|
|
95
|
+
|
|
96
|
+
@classmethod
|
|
97
|
+
def from_env(cls) -> DbSettings:
|
|
98
|
+
return cls(
|
|
99
|
+
engine=os.getenv("DB_ENGINE", cls.engine).lower(),
|
|
100
|
+
host=os.getenv("DB_HOST", cls.host),
|
|
101
|
+
port=_int(os.getenv("DB_PORT"), cls.port),
|
|
102
|
+
user=os.getenv("DB_USER", cls.user),
|
|
103
|
+
password=os.getenv("DB_PASSWORD", cls.password),
|
|
104
|
+
names=_csv(os.getenv("DB_NAMES")),
|
|
105
|
+
mssql_driver=os.getenv("MSSQL_DRIVER", cls.mssql_driver),
|
|
106
|
+
pg_schemas=_csv(os.getenv("PG_SCHEMAS")) or ["public"],
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
@dataclass
|
|
111
|
+
class Settings:
|
|
112
|
+
neo4j: Neo4jSettings = field(default_factory=Neo4jSettings)
|
|
113
|
+
git: GitSettings = field(default_factory=GitSettings)
|
|
114
|
+
db: DbSettings = field(default_factory=DbSettings)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def load_settings(env_file: str | None = None) -> Settings:
|
|
118
|
+
"""Load settings from the environment, seeding from a .env file if present."""
|
|
119
|
+
load_dotenv(dotenv_path=env_file, override=False)
|
|
120
|
+
return Settings(
|
|
121
|
+
neo4j=Neo4jSettings.from_env(),
|
|
122
|
+
git=GitSettings.from_env(),
|
|
123
|
+
db=DbSettings.from_env(),
|
|
124
|
+
)
|