graphforge-neo4j 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (44) hide show
  1. graphforge/__init__.py +12 -0
  2. graphforge/__main__.py +6 -0
  3. graphforge/cli.py +402 -0
  4. graphforge/core/__init__.py +23 -0
  5. graphforge/core/config.py +124 -0
  6. graphforge/core/cypher.py +155 -0
  7. graphforge/core/neo4j_writer.py +230 -0
  8. graphforge/db/__init__.py +66 -0
  9. graphforge/db/base.py +313 -0
  10. graphforge/db/ingest.py +199 -0
  11. graphforge/db/mssql.py +85 -0
  12. graphforge/db/mysql.py +80 -0
  13. graphforge/db/postgres.py +58 -0
  14. graphforge/db/vds.py +137 -0
  15. graphforge/git/__init__.py +5 -0
  16. graphforge/git/clone.py +112 -0
  17. graphforge/git/discover.py +48 -0
  18. graphforge/git/history.py +206 -0
  19. graphforge/git/ingest.py +442 -0
  20. graphforge/git/parsers/__init__.py +10 -0
  21. graphforge/git/parsers/generic.py +78 -0
  22. graphforge/git/parsers/golang.py +291 -0
  23. graphforge/git/parsers/java.py +148 -0
  24. graphforge/git/parsers/pom.py +54 -0
  25. graphforge/git/parsers/python.py +374 -0
  26. graphforge/git/parsers/typescript.py +387 -0
  27. graphforge/git/scan.py +180 -0
  28. graphforge/link/__init__.py +10 -0
  29. graphforge/link/passes.py +128 -0
  30. graphforge/mcp/__init__.py +5 -0
  31. graphforge/mcp/server.py +694 -0
  32. graphforge/py.typed +0 -0
  33. graphforge/schema/db_schema.cypher +24 -0
  34. graphforge/schema/git_schema.cypher +42 -0
  35. graphforge/schema/vds_schema.cypher +15 -0
  36. graphforge/ui/__init__.py +5 -0
  37. graphforge/ui/dashboard.html +731 -0
  38. graphforge/ui/server.py +412 -0
  39. graphforge_neo4j-0.2.0.dist-info/METADATA +541 -0
  40. graphforge_neo4j-0.2.0.dist-info/RECORD +44 -0
  41. graphforge_neo4j-0.2.0.dist-info/WHEEL +5 -0
  42. graphforge_neo4j-0.2.0.dist-info/entry_points.txt +2 -0
  43. graphforge_neo4j-0.2.0.dist-info/licenses/LICENSE +21 -0
  44. graphforge_neo4j-0.2.0.dist-info/top_level.txt +1 -0
graphforge/__init__.py ADDED
@@ -0,0 +1,12 @@
1
+ """graphforge — forge Git repositories and relational databases into a Neo4j knowledge graph.
2
+
3
+ Published on PyPI as ``graphforge-neo4j``; imported as ``graphforge``.
4
+ """
5
+ from __future__ import annotations
6
+
7
+ try:
8
+ from importlib.metadata import version
9
+
10
+ __version__ = version("graphforge-neo4j")
11
+ except ImportError: # not installed (source/editable/dev) — fall back to the literal
12
+ __version__ = "0.1.0"
graphforge/__main__.py ADDED
@@ -0,0 +1,6 @@
1
+ import sys
2
+
3
+ from .cli import main
4
+
5
+ if __name__ == "__main__":
6
+ sys.exit(main())
graphforge/cli.py ADDED
@@ -0,0 +1,402 @@
1
+ """graphforge command-line interface.
2
+
3
+ graphforge init # create graph constraints & indexes
4
+ graphforge git [...] # ingest git repositories (structure + history)
5
+ graphforge db [...] # ingest relational database schemas
6
+ graphforge mcp # serve the graph over MCP
7
+ graphforge verify # connect and report node counts per label
8
+ """
9
+ from __future__ import annotations
10
+
11
+ import argparse
12
+ import json
13
+ import logging
14
+ import os
15
+ import sys
16
+
17
+ from . import __version__
18
+ from .core.config import Settings, load_settings
19
+ from .core.neo4j_writer import Neo4jWriter, load_schema
20
+
21
+
22
+ def _add_common(p: argparse.ArgumentParser) -> None:
23
+ p.add_argument("--env", metavar="FILE", help="path to a .env file to load")
24
+ p.add_argument("--neo4j-uri", dest="neo4j_uri")
25
+ p.add_argument("--neo4j-user", dest="neo4j_user")
26
+ p.add_argument("--neo4j-password", dest="neo4j_password")
27
+ p.add_argument("--neo4j-database", dest="neo4j_database")
28
+
29
+
30
+ def _add_writer_opts(p: argparse.ArgumentParser) -> None:
31
+ p.add_argument("--emit", metavar="FILE",
32
+ help="write a replayable .cypher script instead of pushing to Neo4j")
33
+ p.add_argument("--dry-run", action="store_true",
34
+ help="build operations but neither connect nor write (just count)")
35
+ p.add_argument("--no-schema", action="store_true",
36
+ help="skip creating constraints/indexes before ingesting")
37
+
38
+
39
+ def _settings(args) -> Settings:
40
+ s = load_settings(getattr(args, "env", None))
41
+ if getattr(args, "neo4j_uri", None):
42
+ s.neo4j.uri = args.neo4j_uri
43
+ if getattr(args, "neo4j_user", None):
44
+ s.neo4j.user = args.neo4j_user
45
+ if getattr(args, "neo4j_password", None):
46
+ s.neo4j.password = args.neo4j_password
47
+ if getattr(args, "neo4j_database", None):
48
+ s.neo4j.database = args.neo4j_database
49
+ return s
50
+
51
+
52
+ def _writer(s: Settings, args) -> Neo4jWriter:
53
+ return Neo4jWriter(s.neo4j, emit_path=getattr(args, "emit", None),
54
+ dry_run=getattr(args, "dry_run", False))
55
+
56
+
57
+ def _report(writer: Neo4jWriter, extra: str = "") -> None:
58
+ where = {
59
+ "emit": f"wrote {writer.ops_written} operations -> {writer.emit_path}",
60
+ "dry-run": f"dry run: {writer.ops_written} operations (nothing written)",
61
+ "push": f"pushed {writer.ops_written} operations to Neo4j",
62
+ }[writer.mode]
63
+ print(f"[graphforge] {where}{(' | ' + extra) if extra else ''}")
64
+
65
+
66
+ # ----------------------------------------------------------------------------
67
+ # git
68
+ # ----------------------------------------------------------------------------
69
+ def _resolve_git_sources(args, git_settings) -> list[dict]:
70
+ specs: list[dict] = []
71
+ if args.config:
72
+ with open(args.config, encoding="utf-8") as fh:
73
+ data = json.load(fh)
74
+ default_branch = data.get("defaultBranch", git_settings.default_branch)
75
+ for repo in data.get("repositories", []):
76
+ repo.setdefault("branch", default_branch)
77
+ specs.append(repo)
78
+ for item in args.sources:
79
+ if "://" in item or item.endswith(".git"):
80
+ specs.append({"url": item, "name": args.name, "branch": args.branch})
81
+ else:
82
+ specs.append({"path": item, "name": args.name})
83
+ if not specs and git_settings.gitlab_server and git_settings.gitlab_group_id:
84
+ from .git.discover import gitlab_group_repos
85
+ specs = gitlab_group_repos(
86
+ git_settings.gitlab_server, git_settings.gitlab_group_id,
87
+ git_settings.gitlab_token, git_settings.default_branch,
88
+ since_days=getattr(args, "since", 0) or 0,
89
+ )
90
+ return specs
91
+
92
+
93
+ def cmd_git(args) -> int:
94
+ from .git.ingest import GitIngestor
95
+
96
+ s = _settings(args)
97
+ specs = _resolve_git_sources(args, s.git)
98
+ if not specs:
99
+ print("nothing to ingest: pass a path/URL, -c config.json, or set GITLAB_* env", file=sys.stderr)
100
+ return 2
101
+ with _writer(s, args) as w:
102
+ gi = GitIngestor(w, s.git)
103
+ if not args.no_schema:
104
+ gi.apply_schema()
105
+ stats = gi.ingest(
106
+ specs,
107
+ include_lines=args.lines or s.neo4j.include_lines,
108
+ with_structure=not args.no_structure,
109
+ with_history=not args.no_history,
110
+ replace=args.replace,
111
+ since_commit=getattr(args, "since_commit", "") or "",
112
+ )
113
+ extra = f"{stats['repos']} repos, {stats['files']} files, {stats['commits']} commits"
114
+ if stats.get("failed"):
115
+ extra += f", {stats['failed']} failed (see log with -v)"
116
+ _report(w, extra)
117
+ return 0
118
+
119
+
120
+ # ----------------------------------------------------------------------------
121
+ # db
122
+ # ----------------------------------------------------------------------------
123
+ def _default_schemas(engine: str, db_settings) -> list[str]:
124
+ """PG_SCHEMAS is a PostgreSQL default; other engines discover their own."""
125
+ return list(db_settings.pg_schemas) if (engine or "").lower().startswith(("pg", "postgres")) else []
126
+
127
+
128
+ def _resolve_db_sources(args, db_settings) -> list[dict]:
129
+ sample_rows = getattr(args, "sample_rows", 0) or 0
130
+ if args.config:
131
+ with open(args.config, encoding="utf-8") as fh:
132
+ data = json.load(fh)
133
+ sources = data.get("sources", [])
134
+ for src in sources:
135
+ if "password" not in src and src.get("passwordEnv"):
136
+ src["password"] = os.getenv(src["passwordEnv"], "")
137
+ if sample_rows:
138
+ src["sampleRows"] = sample_rows
139
+ return sources
140
+
141
+ url = getattr(args, "url", None) or os.getenv("DB_URL")
142
+ if url:
143
+ from .db import parse_db_url
144
+ src = parse_db_url(url)
145
+ if args.databases:
146
+ src["databases"] = [d.strip() for d in args.databases.split(",") if d.strip()]
147
+ if args.driver:
148
+ src["driver"] = args.driver
149
+ if args.schemas:
150
+ src["schemas"] = [s.strip() for s in args.schemas.split(",") if s.strip()]
151
+ src["sampleRows"] = sample_rows
152
+ return [src]
153
+
154
+ engine = args.engine or db_settings.engine
155
+ databases = (args.databases.split(",") if args.databases else db_settings.names)
156
+ return [{
157
+ "engine": engine,
158
+ "host": args.host or db_settings.host,
159
+ "port": args.port or db_settings.port,
160
+ "user": args.user or db_settings.user,
161
+ "password": args.password or db_settings.password,
162
+ # empty -> auto-discover every non-system database on the server
163
+ "databases": [d.strip() for d in databases if d.strip()],
164
+ "driver": args.driver or db_settings.mssql_driver,
165
+ "schemas": (args.schemas.split(",") if args.schemas
166
+ else _default_schemas(engine, db_settings)),
167
+ "sampleRows": sample_rows,
168
+ }]
169
+
170
+
171
+ def cmd_db(args) -> int:
172
+ from .db import DbIngestor
173
+
174
+ have_conn = bool(getattr(args, "url", None) or args.config or args.host
175
+ or os.getenv("DB_URL") or os.getenv("DB_HOST"))
176
+ if not have_conn:
177
+ print("provide a connection: --url <db-url>, or --host …, or -c config.json "
178
+ "(or set DB_URL / DB_HOST)", file=sys.stderr)
179
+ return 2
180
+ s = _settings(args)
181
+ sources = _resolve_db_sources(args, s.db)
182
+ with _writer(s, args) as w:
183
+ di = DbIngestor(w)
184
+ if not args.no_schema:
185
+ di.apply_schema()
186
+ stats = di.ingest_sources(sources, replace=args.replace)
187
+ extra = f"{stats['databases']} databases, {stats['tables']} tables, {stats['columns']} columns"
188
+ if stats.get("failed"):
189
+ extra += f", {stats['failed']} failed (see log with -v)"
190
+ _report(w, extra)
191
+ return 0
192
+
193
+
194
+ # ----------------------------------------------------------------------------
195
+ # init / verify / mcp
196
+ # ----------------------------------------------------------------------------
197
+ def cmd_init(args) -> int:
198
+ s = _settings(args)
199
+ with _writer(s, args) as w:
200
+ n = w.apply_schema(load_schema("git_schema.cypher"))
201
+ n += w.apply_schema(load_schema("db_schema.cypher"))
202
+ print(f"[graphforge] applied {n} schema statements ({w.mode})")
203
+ return 0
204
+
205
+
206
+ def cmd_verify(args) -> int:
207
+ s = _settings(args)
208
+ with Neo4jWriter(s.neo4j) as w:
209
+ counts = w.label_counts()
210
+ if not counts:
211
+ print("[graphforge] no nodes found (is the graph empty, or wrong database?)")
212
+ return 0
213
+ print("[graphforge] node counts by label:")
214
+ for label, count in sorted(counts.items(), key=lambda kv: -kv[1]):
215
+ print(f" {label:20s} {count}")
216
+ return 0
217
+
218
+
219
+ def cmd_link(args) -> int:
220
+ from .link import PASSES, LinkRunner
221
+
222
+ selected = [name for name in PASSES if getattr(args, name.replace("-", "_"))]
223
+ if not selected:
224
+ selected = list(PASSES) # default: run every pass
225
+ s = _settings(args)
226
+ with _writer(s, args) as w:
227
+ n = LinkRunner(w).run(selected, min_name_len=args.min_table_name_len)
228
+ _report(w, f"{n} link passes: {', '.join(selected)}")
229
+ return 0
230
+
231
+
232
+ def cmd_vds(args) -> int:
233
+ from .db.vds import VdsIngestor
234
+
235
+ s = _settings(args)
236
+ database = args.database or (s.db.names[0] if s.db.names else "")
237
+ if not database:
238
+ print("VDS import needs --database (or DB_NAMES)", file=sys.stderr)
239
+ return 2
240
+ with _writer(s, args) as w:
241
+ vi = VdsIngestor(w)
242
+ if not args.no_schema:
243
+ vi.apply_schema()
244
+ stats = vi.ingest(
245
+ engine=args.engine or s.db.engine, host=args.host or s.db.host,
246
+ port=args.port or s.db.port, user=args.user or s.db.user,
247
+ password=args.password or s.db.password, database=database,
248
+ driver=args.driver or s.db.mssql_driver,
249
+ )
250
+ _report(w, f"{stats['services']} VDS services, {stats['queries']} queries, "
251
+ f"{stats['whereFields']} where-fields")
252
+ return 0
253
+
254
+
255
+ def cmd_status(args) -> int:
256
+ s = _settings(args)
257
+ with Neo4jWriter(s.neo4j) as w:
258
+ rows = w.repository_status()
259
+ if not rows:
260
+ print("[graphforge] no repositories ingested yet")
261
+ return 0
262
+ print(f"{'repository':30s} {'status':12s} {'files':>7s} {'commits':>8s} lastIngestedAt")
263
+ for r in rows:
264
+ print(f"{(r.get('name') or ''):30s} {(r.get('status') or ''):12s} "
265
+ f"{(r.get('files') or 0):>7} {(r.get('commits') or 0):>8} {r.get('lastIngestedAt') or ''}")
266
+ return 0
267
+
268
+
269
+ def cmd_ui(args) -> int:
270
+ from .ui import serve
271
+
272
+ s = _settings(args)
273
+ serve(s, host=args.host, port=args.port)
274
+ return 0
275
+
276
+
277
+ def cmd_mcp(args) -> int:
278
+ from .mcp.server import run
279
+
280
+ s = _settings(args)
281
+ run(s.neo4j)
282
+ return 0
283
+
284
+
285
+ # ----------------------------------------------------------------------------
286
+ def build_parser() -> argparse.ArgumentParser:
287
+ from .link import passes as link_passes # for the --min-table-name-len default
288
+
289
+ parser = argparse.ArgumentParser(prog="graphforge", description=__doc__,
290
+ formatter_class=argparse.RawDescriptionHelpFormatter)
291
+ parser.add_argument("--version", action="version", version=f"graphforge {__version__}")
292
+ parser.add_argument("-v", "--verbose", action="store_true")
293
+ sub = parser.add_subparsers(dest="command", required=True)
294
+
295
+ p_init = sub.add_parser("init", help="create graph constraints and indexes")
296
+ _add_common(p_init)
297
+ _add_writer_opts(p_init)
298
+ p_init.set_defaults(func=cmd_init)
299
+
300
+ p_git = sub.add_parser("git", help="ingest git repositories (code structure + history)")
301
+ p_git.add_argument("sources", nargs="*", help="local repo paths and/or git URLs")
302
+ p_git.add_argument("-c", "--config", help="repositories.json")
303
+ p_git.add_argument("--name", help="name for a single repo passed positionally")
304
+ p_git.add_argument("--branch", help="branch for a single URL")
305
+ p_git.add_argument("--lines", action="store_true", help="store per-line :Line nodes (heavy)")
306
+ p_git.add_argument("--no-structure", action="store_true", help="skip code structure")
307
+ p_git.add_argument("--no-history", action="store_true", help="skip commit history")
308
+ p_git.add_argument("--replace", action="store_true", help="delete each repo's existing subgraph first")
309
+ p_git.add_argument("--since", type=int, default=0, metavar="DAYS",
310
+ help="GitLab discovery: only repos active in the last N days")
311
+ p_git.add_argument("--since-commit", dest="since_commit", metavar="SHA", default="",
312
+ help="incremental ingest: only commits after SHA (or 'auto' to "
313
+ "continue from :Repository.lastCommit)")
314
+ _add_common(p_git)
315
+ _add_writer_opts(p_git)
316
+ p_git.set_defaults(func=cmd_git)
317
+
318
+ p_db = sub.add_parser("db", help="ingest relational database schemas")
319
+ p_db.add_argument("-c", "--config", help="databases.json")
320
+ p_db.add_argument("--url", help="connection URL, e.g. postgresql://user:pass@host:5432/dbname "
321
+ "(omit the database to graph every non-system database)")
322
+ p_db.add_argument("--engine", help="mysql | postgres | mssql")
323
+ p_db.add_argument("--host")
324
+ p_db.add_argument("--port", type=int)
325
+ p_db.add_argument("--user")
326
+ p_db.add_argument("--password")
327
+ p_db.add_argument("--databases", help="comma-separated database names")
328
+ p_db.add_argument("--driver", help="MSSQL ODBC driver name")
329
+ p_db.add_argument("--schemas", help="comma-separated schema filter (postgres/mssql)")
330
+ p_db.add_argument("--sample-rows", dest="sample_rows", type=int, default=0, metavar="N",
331
+ help="opt-in profiling: COUNT(*) every table into :Table.approxRows, "
332
+ "and for tables of at most N rows also COUNT(DISTINCT col) into "
333
+ ":Column.approxCardinality (default 0 = off)")
334
+ p_db.add_argument("--replace", action="store_true", help="delete each database's existing subgraph first")
335
+ _add_common(p_db)
336
+ _add_writer_opts(p_db)
337
+ p_db.set_defaults(func=cmd_db)
338
+
339
+ p_vds = sub.add_parser("vds", help="import a VDS (virtual data service) catalog")
340
+ p_vds.add_argument("--engine", help="mysql | postgres | mssql")
341
+ p_vds.add_argument("--host")
342
+ p_vds.add_argument("--port", type=int)
343
+ p_vds.add_argument("--user")
344
+ p_vds.add_argument("--password")
345
+ p_vds.add_argument("--database", help="database holding the VDS catalog tables")
346
+ p_vds.add_argument("--driver", help="MSSQL ODBC driver name")
347
+ _add_common(p_vds)
348
+ _add_writer_opts(p_vds)
349
+ p_vds.set_defaults(func=cmd_vds)
350
+
351
+ p_link = sub.add_parser("link", help="create code<->database links over the loaded graph")
352
+ p_link.add_argument("--maps-to", action="store_true", help="JPA entity -> Table")
353
+ p_link.add_argument("--based-on", action="store_true", help="View -> Table (same DB)")
354
+ p_link.add_argument("--uses-table", action="store_true", help="StoredProcedure -> Table (same DB)")
355
+ p_link.add_argument("--cross-db", action="store_true", help="View -> Table (different DB)")
356
+ p_link.add_argument("--min-table-name-len", dest="min_table_name_len", type=int,
357
+ default=link_passes.DEFAULT_MIN_NAME_LEN, metavar="N",
358
+ help="ignore table names shorter than N characters in the SQL-text "
359
+ f"passes (default {link_passes.DEFAULT_MIN_NAME_LEN})")
360
+ _add_common(p_link)
361
+ _add_writer_opts(p_link)
362
+ p_link.set_defaults(func=cmd_link)
363
+
364
+ p_verify = sub.add_parser("verify", help="report node counts per label")
365
+ _add_common(p_verify)
366
+ p_verify.set_defaults(func=cmd_verify)
367
+
368
+ p_status = sub.add_parser("status", help="show per-repository ingest status")
369
+ _add_common(p_status)
370
+ p_status.set_defaults(func=cmd_status)
371
+
372
+ p_ui = sub.add_parser("ui", help="serve a local web dashboard (config + load status)")
373
+ p_ui.add_argument("--host", default="127.0.0.1")
374
+ p_ui.add_argument("--port", type=int, default=8000)
375
+ _add_common(p_ui)
376
+ p_ui.set_defaults(func=cmd_ui)
377
+
378
+ p_mcp = sub.add_parser("mcp", help="serve the knowledge graph over MCP (stdio)")
379
+ _add_common(p_mcp)
380
+ p_mcp.set_defaults(func=cmd_mcp)
381
+
382
+ return parser
383
+
384
+
385
+ def main(argv: list[str] | None = None) -> int:
386
+ parser = build_parser()
387
+ args = parser.parse_args(argv)
388
+ logging.basicConfig(
389
+ level=logging.INFO if args.verbose else logging.WARNING,
390
+ format="%(levelname)s %(name)s: %(message)s",
391
+ )
392
+ try:
393
+ return args.func(args)
394
+ except (ValueError, RuntimeError, FileNotFoundError) as exc:
395
+ # Expected, user-facing failures (bad engine, missing driver, missing
396
+ # config file) exit cleanly instead of dumping a traceback.
397
+ print(f"error: {exc}", file=sys.stderr)
398
+ return 2
399
+
400
+
401
+ if __name__ == "__main__":
402
+ sys.exit(main())
@@ -0,0 +1,23 @@
1
+ """Core: configuration, graph operation model, Cypher builders, and the Neo4j writer."""
2
+
3
+ from .cypher import (
4
+ NodeRef,
5
+ Operation,
6
+ escape_cypher_string,
7
+ lit,
8
+ merge_node,
9
+ merge_rel,
10
+ set_label,
11
+ )
12
+ from .neo4j_writer import Neo4jWriter
13
+
14
+ __all__ = [
15
+ "Neo4jWriter",
16
+ "NodeRef",
17
+ "Operation",
18
+ "escape_cypher_string",
19
+ "lit",
20
+ "merge_node",
21
+ "merge_rel",
22
+ "set_label",
23
+ ]
@@ -0,0 +1,124 @@
1
+ """Environment-driven configuration. Nothing here is hardcoded to any host.
2
+
3
+ Values are read from the process environment (optionally seeded from a local
4
+ `.env` file via python-dotenv). CLI flags may override individual fields.
5
+ """
6
+ from __future__ import annotations
7
+
8
+ import os
9
+ from dataclasses import dataclass, field
10
+
11
+ try:
12
+ from dotenv import load_dotenv
13
+ except ImportError: # pragma: no cover - dotenv is a core dep but keep import soft
14
+ def load_dotenv(*_args, **_kwargs): # type: ignore
15
+ return False
16
+
17
+
18
+ def _bool(value: str | None, default: bool = False) -> bool:
19
+ if value is None:
20
+ return default
21
+ return value.strip().lower() in {"1", "true", "yes", "on"}
22
+
23
+
24
+ def _int(value: str | None, default: int) -> int:
25
+ try:
26
+ return int(str(value).strip())
27
+ except (TypeError, ValueError):
28
+ return default
29
+
30
+
31
+ def _csv(value: str | None) -> list[str]:
32
+ if not value:
33
+ return []
34
+ return [item.strip() for item in value.split(",") if item.strip()]
35
+
36
+
37
+ @dataclass
38
+ class Neo4jSettings:
39
+ uri: str = "bolt://127.0.0.1:7687"
40
+ user: str = "neo4j"
41
+ password: str = ""
42
+ database: str = "neo4j"
43
+ batch_size: int = 500
44
+ include_lines: bool = False
45
+
46
+ @classmethod
47
+ def from_env(cls) -> Neo4jSettings:
48
+ return cls(
49
+ uri=os.getenv("NEO4J_URI", cls.uri),
50
+ user=os.getenv("NEO4J_USER", cls.user),
51
+ password=os.getenv("NEO4J_PASSWORD", cls.password),
52
+ database=os.getenv("NEO4J_DATABASE", cls.database),
53
+ batch_size=_int(os.getenv("GF_BATCH_SIZE"), cls.batch_size),
54
+ include_lines=_bool(os.getenv("GF_INCLUDE_LINES"), cls.include_lines),
55
+ )
56
+
57
+
58
+ @dataclass
59
+ class GitSettings:
60
+ repo_dir: str = "./repos"
61
+ username: str = ""
62
+ password: str = ""
63
+ token: str = ""
64
+ gitlab_server: str = ""
65
+ gitlab_group_id: str = ""
66
+ gitlab_token: str = ""
67
+ default_branch: str = "main"
68
+ history_limit: int = 0
69
+
70
+ @classmethod
71
+ def from_env(cls) -> GitSettings:
72
+ return cls(
73
+ repo_dir=os.getenv("GF_REPO_DIR", cls.repo_dir),
74
+ username=os.getenv("GIT_USERNAME", cls.username),
75
+ password=os.getenv("GIT_PASSWORD", cls.password),
76
+ token=os.getenv("GIT_TOKEN", cls.token),
77
+ gitlab_server=os.getenv("GITLAB_SERVER", cls.gitlab_server),
78
+ gitlab_group_id=os.getenv("GITLAB_GROUP_ID", cls.gitlab_group_id),
79
+ gitlab_token=os.getenv("GITLAB_TOKEN", cls.gitlab_token),
80
+ default_branch=os.getenv("GIT_DEFAULT_BRANCH", cls.default_branch),
81
+ history_limit=_int(os.getenv("GF_HISTORY_LIMIT"), cls.history_limit),
82
+ )
83
+
84
+
85
+ @dataclass
86
+ class DbSettings:
87
+ engine: str = "mysql"
88
+ host: str = "127.0.0.1"
89
+ port: int = 3306
90
+ user: str = ""
91
+ password: str = ""
92
+ names: list[str] = field(default_factory=list)
93
+ mssql_driver: str = "ODBC Driver 18 for SQL Server"
94
+ pg_schemas: list[str] = field(default_factory=lambda: ["public"])
95
+
96
+ @classmethod
97
+ def from_env(cls) -> DbSettings:
98
+ return cls(
99
+ engine=os.getenv("DB_ENGINE", cls.engine).lower(),
100
+ host=os.getenv("DB_HOST", cls.host),
101
+ port=_int(os.getenv("DB_PORT"), cls.port),
102
+ user=os.getenv("DB_USER", cls.user),
103
+ password=os.getenv("DB_PASSWORD", cls.password),
104
+ names=_csv(os.getenv("DB_NAMES")),
105
+ mssql_driver=os.getenv("MSSQL_DRIVER", cls.mssql_driver),
106
+ pg_schemas=_csv(os.getenv("PG_SCHEMAS")) or ["public"],
107
+ )
108
+
109
+
110
+ @dataclass
111
+ class Settings:
112
+ neo4j: Neo4jSettings = field(default_factory=Neo4jSettings)
113
+ git: GitSettings = field(default_factory=GitSettings)
114
+ db: DbSettings = field(default_factory=DbSettings)
115
+
116
+
117
+ def load_settings(env_file: str | None = None) -> Settings:
118
+ """Load settings from the environment, seeding from a .env file if present."""
119
+ load_dotenv(dotenv_path=env_file, override=False)
120
+ return Settings(
121
+ neo4j=Neo4jSettings.from_env(),
122
+ git=GitSettings.from_env(),
123
+ db=DbSettings.from_env(),
124
+ )