ripple-sql 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ripple/__init__.py +31 -0
- ripple/answer.py +473 -0
- ripple/answer_page.py +214 -0
- ripple/cache.py +80 -0
- ripple/ci.py +422 -0
- ripple/ci_signature.py +374 -0
- ripple/cli.py +733 -0
- ripple/doctor.py +225 -0
- ripple/engine/__init__.py +111 -0
- ripple/engine/budget.py +86 -0
- ripple/engine/column_lineage.py +112 -0
- ripple/engine/column_ref.py +818 -0
- ripple/engine/cte_tracing.py +1309 -0
- ripple/engine/dependencies.py +466 -0
- ripple/engine/dialect.py +132 -0
- ripple/engine/dispatch.py +12 -0
- ripple/engine/extraction.py +27 -0
- ripple/engine/jinja.py +282 -0
- ripple/engine/json_sources.py +241 -0
- ripple/engine/macro_source.py +127 -0
- ripple/engine/pipeline.py +265 -0
- ripple/engine/preprocess.py +174 -0
- ripple/engine/safe_gen.py +21 -0
- ripple/engine/schema_qualification.py +151 -0
- ripple/engine/scope.py +488 -0
- ripple/engine/select_sources.py +1038 -0
- ripple/engine/sql_script.py +729 -0
- ripple/engine/statement.py +449 -0
- ripple/engine/tech_debt.py +169 -0
- ripple/engine/tsql_catalog.py +83 -0
- ripple/engine/tsql_scalar_vars.py +248 -0
- ripple/engine/tsql_tvf.py +653 -0
- ripple/engine/tsql_xml.py +97 -0
- ripple/engine/types.py +167 -0
- ripple/engine/unused_deps.py +555 -0
- ripple/engine/validation.py +158 -0
- ripple/graph.py +1499 -0
- ripple/home.py +232 -0
- ripple/loaders/__init__.py +7 -0
- ripple/loaders/dbt.py +359 -0
- ripple/loaders/dbt_config.py +339 -0
- ripple/loaders/identity.py +328 -0
- ripple/loaders/sidecar.py +65 -0
- ripple/loaders/sqldir.py +262 -0
- ripple/loaders/types.py +197 -0
- ripple/lookml.py +163 -0
- ripple/mcp_server.py +600 -0
- ripple/names.py +40 -0
- ripple/project.py +167 -0
- ripple/py.typed +0 -0
- ripple/render.py +426 -0
- ripple/render_shims.py +209 -0
- ripple/schemas.py +155 -0
- ripple/semantic.py +232 -0
- ripple/server.py +184 -0
- ripple/sourcefiles.py +64 -0
- ripple/star_resolution.py +100 -0
- ripple/static/answer.css +146 -0
- ripple/static/answer.html +358 -0
- ripple/static/answer_twin.js +299 -0
- ripple/static/explore.js +133 -0
- ripple/usage/__init__.py +18 -0
- ripple/usage/cli.py +78 -0
- ripple/usage/collect.py +315 -0
- ripple/usage/discover.py +190 -0
- ripple/usage/ingest.py +414 -0
- ripple/usage/report.py +131 -0
- ripple_sql-0.1.0.dist-info/METADATA +285 -0
- ripple_sql-0.1.0.dist-info/RECORD +72 -0
- ripple_sql-0.1.0.dist-info/WHEEL +4 -0
- ripple_sql-0.1.0.dist-info/entry_points.txt +3 -0
- ripple_sql-0.1.0.dist-info/licenses/LICENSE +202 -0
ripple/usage/collect.py
ADDED
|
@@ -0,0 +1,315 @@
|
|
|
1
|
+
"""Run the user's already-authenticated warehouse CLI and ingest the result.
|
|
2
|
+
|
|
3
|
+
Ripple never opens a warehouse connection of its own: the child process is
|
|
4
|
+
`snow`, `bq`, or `databricks`, using whatever auth the user set up for it.
|
|
5
|
+
The rows land in a temp file that is aggregated and deleted inside one call,
|
|
6
|
+
so query text never reaches the caller (human or model) and is never stored.
|
|
7
|
+
|
|
8
|
+
Validation is strict on purpose: a clean exit code AND stdout that parses as
|
|
9
|
+
complete JSON of the expected shape, or the whole collect is refused. Partial
|
|
10
|
+
output silently ingested is how usage numbers end up confidently wrong.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
import json
|
|
16
|
+
import re
|
|
17
|
+
import subprocess
|
|
18
|
+
import tempfile
|
|
19
|
+
import time
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
|
|
22
|
+
from ripple.usage import discover
|
|
23
|
+
from ripple.usage.ingest import ingest_file
|
|
24
|
+
|
|
25
|
+
DEFAULT_TIMEOUT = 60.0
|
|
26
|
+
SNOWFLAKE_INFOSCHEMA_MAX_DAYS = 7
|
|
27
|
+
ACCOUNT_ROW_LIMIT = 100_000
|
|
28
|
+
DATABRICKS_PAGE_SIZE = 1000
|
|
29
|
+
|
|
30
|
+
SNOWFLAKE_GRANT_HINT = (
|
|
31
|
+
"Reading everyone's queries needs one admin grant. Have an admin run:\n"
|
|
32
|
+
" GRANT IMPORTED PRIVILEGES ON DATABASE SNOWFLAKE TO ROLE <your role>;\n"
|
|
33
|
+
"or rerun with scope='mine' (your own queries, no permission needed)."
|
|
34
|
+
)
|
|
35
|
+
BIGQUERY_GRANT_HINT = (
|
|
36
|
+
"Reading everyone's jobs needs bigquery.jobs.listAll on the project "
|
|
37
|
+
"(roles/bigquery.resourceViewer). Or rerun with scope='mine'."
|
|
38
|
+
)
|
|
39
|
+
|
|
40
|
+
_COLUMNS = "query_id, query_text, database_name, schema_name, user_name"
|
|
41
|
+
_SNOW_TIME = "TO_VARCHAR(start_time, 'YYYY-MM-DD\"T\"HH24:MI:SSTZH:TZM') AS start_time"
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
class CollectError(Exception):
|
|
45
|
+
"""A refusal with the reason and, when known, the exact fix."""
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _snowflake_sql(days: int, scope: str) -> str:
|
|
49
|
+
if scope == "account":
|
|
50
|
+
return (
|
|
51
|
+
f"SELECT {_COLUMNS}, {_SNOW_TIME} FROM SNOWFLAKE.ACCOUNT_USAGE.QUERY_HISTORY "
|
|
52
|
+
f"WHERE start_time >= DATEADD('day', -{days}, CURRENT_TIMESTAMP()) "
|
|
53
|
+
f"LIMIT {ACCOUNT_ROW_LIMIT}"
|
|
54
|
+
)
|
|
55
|
+
days = min(days, SNOWFLAKE_INFOSCHEMA_MAX_DAYS)
|
|
56
|
+
return (
|
|
57
|
+
f"SELECT {_COLUMNS}, {_SNOW_TIME} FROM TABLE(INFORMATION_SCHEMA.QUERY_HISTORY("
|
|
58
|
+
f"END_TIME_RANGE_START => DATEADD('day', -{days}, CURRENT_TIMESTAMP()), "
|
|
59
|
+
f"RESULT_LIMIT => 10000))"
|
|
60
|
+
)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _bigquery_sql(days: int, scope: str, region: str) -> str:
|
|
64
|
+
view = "JOBS_BY_PROJECT" if scope == "account" else "JOBS_BY_USER"
|
|
65
|
+
return (
|
|
66
|
+
"SELECT job_id AS query_id, query AS query_text, user_email AS user_name, "
|
|
67
|
+
"FORMAT_TIMESTAMP('%Y-%m-%dT%H:%M:%S+00:00', creation_time) AS start_time "
|
|
68
|
+
f"FROM `region-{region}`.INFORMATION_SCHEMA.{view} "
|
|
69
|
+
f"WHERE creation_time > TIMESTAMP_SUB(CURRENT_TIMESTAMP(), INTERVAL {days} DAY) "
|
|
70
|
+
"AND job_type = 'QUERY' AND query IS NOT NULL "
|
|
71
|
+
"ORDER BY creation_time DESC LIMIT 10000"
|
|
72
|
+
)
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _parse_json_stdout(stdout: str, cli: str):
|
|
76
|
+
text = stdout.strip()
|
|
77
|
+
if not text:
|
|
78
|
+
raise CollectError(f"{cli} exited cleanly but printed nothing; nothing was ingested.")
|
|
79
|
+
try:
|
|
80
|
+
return json.loads(text)
|
|
81
|
+
except json.JSONDecodeError as e:
|
|
82
|
+
raise CollectError(
|
|
83
|
+
f"{cli} printed something that is not complete JSON ({e}); refusing to "
|
|
84
|
+
"ingest a partial or polluted export. First bytes: " + text[:120]
|
|
85
|
+
) from None
|
|
86
|
+
|
|
87
|
+
|
|
88
|
+
def _rows_from_array(payload, cli: str) -> list[dict]:
|
|
89
|
+
if isinstance(payload, list) and payload and all(isinstance(r, list) for r in payload):
|
|
90
|
+
payload = [row for block in payload for row in block] # multi-statement output
|
|
91
|
+
if not isinstance(payload, list) or not all(isinstance(r, dict) for r in payload):
|
|
92
|
+
raise CollectError(f"{cli} returned JSON of an unexpected shape (wanted an array of rows).")
|
|
93
|
+
return payload
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def _normalize_snowflake(stdout: str) -> tuple[list[dict], bool]:
|
|
97
|
+
return _rows_from_array(_parse_json_stdout(stdout, "snow"), "snow"), False
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _normalize_bigquery(stdout: str) -> tuple[list[dict], bool]:
|
|
101
|
+
return _rows_from_array(_parse_json_stdout(stdout, "bq"), "bq"), False
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def _normalize_databricks(stdout: str) -> tuple[list[dict], bool]:
|
|
105
|
+
payload = _parse_json_stdout(stdout, "databricks")
|
|
106
|
+
if not isinstance(payload, dict) or not isinstance(payload.get("res"), list):
|
|
107
|
+
raise CollectError(
|
|
108
|
+
"databricks returned JSON without a 'res' list (wanted the "
|
|
109
|
+
"/api/2.0/sql/history/queries response)."
|
|
110
|
+
)
|
|
111
|
+
rows = []
|
|
112
|
+
for q in payload["res"]:
|
|
113
|
+
if not isinstance(q, dict):
|
|
114
|
+
continue
|
|
115
|
+
rows.append(
|
|
116
|
+
{
|
|
117
|
+
"query_id": q.get("query_id"),
|
|
118
|
+
"query_text": q.get("query_text"),
|
|
119
|
+
"user_name": q.get("user_name"),
|
|
120
|
+
"start_time": q.get("query_start_time_ms"),
|
|
121
|
+
}
|
|
122
|
+
)
|
|
123
|
+
return rows, bool(payload.get("has_next_page"))
|
|
124
|
+
|
|
125
|
+
|
|
126
|
+
def resolve_connection(platform: str, requested: str | None) -> tuple[str | None, str]:
|
|
127
|
+
"""(connection to pass, how it was chosen). Refuses to pick among several.
|
|
128
|
+
|
|
129
|
+
An explicit request always wins. Otherwise only the user's own declared
|
|
130
|
+
default (default_connection_name, the active gcloud config, a DEFAULT
|
|
131
|
+
profile) or a single unambiguous entry is used; several candidates with
|
|
132
|
+
no default is an error that lists them, never a silent pick.
|
|
133
|
+
"""
|
|
134
|
+
if requested:
|
|
135
|
+
return requested, "requested"
|
|
136
|
+
if platform == "snowflake":
|
|
137
|
+
found = discover.snowflake_connections()
|
|
138
|
+
names = sorted(found["connections"])
|
|
139
|
+
if found["default"]:
|
|
140
|
+
return found["default"], "default_connection_name"
|
|
141
|
+
if len(names) == 1:
|
|
142
|
+
return names[0], "only connection configured"
|
|
143
|
+
if not names:
|
|
144
|
+
return None, "no connections.toml entries; snow's own default applies"
|
|
145
|
+
raise CollectError(
|
|
146
|
+
f"Several Snowflake connections are configured ({', '.join(names)}) and none "
|
|
147
|
+
"is marked default. Pass connection=<name>; picking one silently could run "
|
|
148
|
+
"against the wrong account or role."
|
|
149
|
+
)
|
|
150
|
+
if platform == "bigquery":
|
|
151
|
+
found = discover.bigquery_configurations()
|
|
152
|
+
active = found["active"]
|
|
153
|
+
project = found["configurations"].get(active) if active else None
|
|
154
|
+
if project:
|
|
155
|
+
return None, f"active gcloud configuration '{active}' (project {project})"
|
|
156
|
+
return None, "bq's own default project applies"
|
|
157
|
+
if platform == "databricks":
|
|
158
|
+
profiles = discover.databricks_profiles()
|
|
159
|
+
names = sorted(profiles)
|
|
160
|
+
if "DEFAULT" in profiles:
|
|
161
|
+
return None, "DEFAULT profile"
|
|
162
|
+
if len(names) == 1:
|
|
163
|
+
return names[0], "only profile configured"
|
|
164
|
+
if not names:
|
|
165
|
+
return None, "no ~/.databrickscfg profiles; the CLI's own auth applies"
|
|
166
|
+
raise CollectError(
|
|
167
|
+
f"Several Databricks profiles are configured ({', '.join(names)}) and none is "
|
|
168
|
+
"DEFAULT. Pass connection=<profile>; picking one silently could run against "
|
|
169
|
+
"the wrong workspace."
|
|
170
|
+
)
|
|
171
|
+
raise CollectError(f"unknown platform '{platform}' (snowflake, bigquery, or databricks)")
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def _build_command(
|
|
175
|
+
platform: str, connection: str | None, days: int, scope: str, region: str
|
|
176
|
+
) -> list[str]:
|
|
177
|
+
if platform == "snowflake":
|
|
178
|
+
argv = ["snow", "sql", "-q", _snowflake_sql(days, scope), "--format", "json"]
|
|
179
|
+
if connection:
|
|
180
|
+
argv += ["-c", connection]
|
|
181
|
+
return argv
|
|
182
|
+
if platform == "bigquery":
|
|
183
|
+
argv = ["bq", "query", "--nouse_legacy_sql", "--format=json", "--max_rows=10000"]
|
|
184
|
+
if connection:
|
|
185
|
+
argv += ["--project_id", connection]
|
|
186
|
+
return argv + [_bigquery_sql(days, scope, region)]
|
|
187
|
+
if platform == "databricks":
|
|
188
|
+
body = {
|
|
189
|
+
"max_results": DATABRICKS_PAGE_SIZE,
|
|
190
|
+
"filter_by": {
|
|
191
|
+
"query_start_time_range": {
|
|
192
|
+
"start_time_ms": int((time.time() - days * 86400) * 1000)
|
|
193
|
+
}
|
|
194
|
+
},
|
|
195
|
+
}
|
|
196
|
+
argv = ["databricks", "api", "post", "/api/2.0/sql/history/queries"]
|
|
197
|
+
if connection:
|
|
198
|
+
argv += ["--profile", connection]
|
|
199
|
+
return argv + ["--json", json.dumps(body)]
|
|
200
|
+
raise CollectError(f"unknown platform '{platform}' (snowflake, bigquery, or databricks)")
|
|
201
|
+
|
|
202
|
+
|
|
203
|
+
NORMALIZERS = {
|
|
204
|
+
"snowflake": _normalize_snowflake,
|
|
205
|
+
"bigquery": _normalize_bigquery,
|
|
206
|
+
"databricks": _normalize_databricks,
|
|
207
|
+
}
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def _permission_hint(platform: str, scope: str, stderr: str) -> str:
|
|
211
|
+
lowered = stderr.lower()
|
|
212
|
+
if scope != "account":
|
|
213
|
+
return ""
|
|
214
|
+
if platform == "snowflake" and ("not authorized" in lowered or "does not exist" in lowered):
|
|
215
|
+
return "\n" + SNOWFLAKE_GRANT_HINT
|
|
216
|
+
if platform == "bigquery" and ("access denied" in lowered or "permission" in lowered):
|
|
217
|
+
return "\n" + BIGQUERY_GRANT_HINT
|
|
218
|
+
return ""
|
|
219
|
+
|
|
220
|
+
|
|
221
|
+
def _run_cli(argv: list[str], timeout: float, platform: str, scope: str) -> str:
|
|
222
|
+
try:
|
|
223
|
+
proc = subprocess.run(argv, capture_output=True, text=True, timeout=timeout)
|
|
224
|
+
except FileNotFoundError:
|
|
225
|
+
raise CollectError(
|
|
226
|
+
f"{argv[0]} is not installed (or not on PATH). Install and authenticate it "
|
|
227
|
+
"once, then collect works with no further setup."
|
|
228
|
+
) from None
|
|
229
|
+
except subprocess.TimeoutExpired:
|
|
230
|
+
raise CollectError(
|
|
231
|
+
f"{argv[0]} did not finish within {timeout:.0f}s. If it opened a browser "
|
|
232
|
+
"window to sign in, finish signing in there and run collect again."
|
|
233
|
+
) from None
|
|
234
|
+
if proc.returncode != 0:
|
|
235
|
+
tail = (proc.stderr or proc.stdout or "").strip()[-600:]
|
|
236
|
+
raise CollectError(
|
|
237
|
+
f"{argv[0]} exited with code {proc.returncode}; nothing was ingested.\n{tail}"
|
|
238
|
+
+ _permission_hint(platform, scope, tail)
|
|
239
|
+
)
|
|
240
|
+
return proc.stdout
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def collect(
|
|
244
|
+
platform: str,
|
|
245
|
+
model_names: list[str],
|
|
246
|
+
connection: str | None = None,
|
|
247
|
+
days: int = 7,
|
|
248
|
+
scope: str = "mine",
|
|
249
|
+
region: str = "us",
|
|
250
|
+
timeout: float = DEFAULT_TIMEOUT,
|
|
251
|
+
model_aliases: dict[str, set[str]] | None = None,
|
|
252
|
+
) -> dict:
|
|
253
|
+
"""One collect: resolve connection, run the CLI, validate, ingest, clean up.
|
|
254
|
+
|
|
255
|
+
Returns the usage store (aggregates only). Raises CollectError with the
|
|
256
|
+
reason and the fix; a failed collect never half-writes anything.
|
|
257
|
+
"""
|
|
258
|
+
if platform not in NORMALIZERS:
|
|
259
|
+
raise CollectError(f"unknown platform '{platform}' (snowflake, bigquery, or databricks)")
|
|
260
|
+
if scope not in ("mine", "account"):
|
|
261
|
+
raise CollectError(f"unknown scope '{scope}' (mine or account)")
|
|
262
|
+
if not re.fullmatch(r"[A-Za-z0-9-]+", region or ""):
|
|
263
|
+
# the region lands inside a backtick-quoted identifier; anything
|
|
264
|
+
# beyond a plain region name could smuggle SQL into the user's
|
|
265
|
+
# authenticated bq session
|
|
266
|
+
raise CollectError(f"'{region}' is not a BigQuery region name (letters, digits, hyphens)")
|
|
267
|
+
requested_days = max(1, int(days))
|
|
268
|
+
days = requested_days
|
|
269
|
+
if platform == "snowflake" and scope == "mine":
|
|
270
|
+
# INFORMATION_SCHEMA retains 7 days; asking for more must not
|
|
271
|
+
# silently claim a window the query never covered
|
|
272
|
+
days = min(days, SNOWFLAKE_INFOSCHEMA_MAX_DAYS)
|
|
273
|
+
connection, chosen_by = resolve_connection(platform, connection)
|
|
274
|
+
argv = _build_command(platform, connection, days, scope, region)
|
|
275
|
+
stdout = _run_cli(argv, timeout, platform, scope)
|
|
276
|
+
records, cli_truncated = NORMALIZERS[platform](stdout)
|
|
277
|
+
if not records:
|
|
278
|
+
window = f"{days} day(s)" + (
|
|
279
|
+
f" (clamped from {requested_days}; INFORMATION_SCHEMA keeps 7 days)"
|
|
280
|
+
if days != requested_days
|
|
281
|
+
else ""
|
|
282
|
+
)
|
|
283
|
+
raise CollectError(
|
|
284
|
+
f"The export came back empty: no queries in the last {window} for this "
|
|
285
|
+
"connection and scope. Try a longer window (days=...) or scope='account'."
|
|
286
|
+
)
|
|
287
|
+
with tempfile.TemporaryDirectory(prefix="ripple-collect-") as tmp:
|
|
288
|
+
spool = Path(tmp) / f"{platform}-history.jsonl"
|
|
289
|
+
spool.write_text(
|
|
290
|
+
"".join(json.dumps(r, default=str) + "\n" for r in records), encoding="utf-8"
|
|
291
|
+
)
|
|
292
|
+
store = ingest_file(spool, model_names, dialect=platform, model_aliases=model_aliases)
|
|
293
|
+
store["manifest"]["collected"] = {
|
|
294
|
+
"platform": platform,
|
|
295
|
+
"connection": connection or "(CLI default)",
|
|
296
|
+
"chosen_by": chosen_by,
|
|
297
|
+
# the Databricks history API has no mine/account split: it returns
|
|
298
|
+
# whatever this login can see, which for an admin is everyone
|
|
299
|
+
"scope": "visible-to-this-login" if platform == "databricks" else scope,
|
|
300
|
+
"days": days,
|
|
301
|
+
"command": " ".join(argv),
|
|
302
|
+
}
|
|
303
|
+
if days != requested_days:
|
|
304
|
+
store["manifest"]["collected"]["requested_days"] = requested_days
|
|
305
|
+
if cli_truncated:
|
|
306
|
+
store["manifest"]["truncated"] = (
|
|
307
|
+
f"yes: the CLI returned its first page of {DATABRICKS_PAGE_SIZE} results and "
|
|
308
|
+
"reported more; the window was cut short"
|
|
309
|
+
)
|
|
310
|
+
elif platform == "snowflake" and scope == "account" and len(records) == ACCOUNT_ROW_LIMIT:
|
|
311
|
+
store["manifest"]["truncated"] = (
|
|
312
|
+
f"likely: exactly {ACCOUNT_ROW_LIMIT:,} rows is the collect limit for "
|
|
313
|
+
"account scope, so the window was probably cut short"
|
|
314
|
+
)
|
|
315
|
+
return store
|
ripple/usage/discover.py
ADDED
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
"""Find the warehouse CLIs already on this machine and their configured
|
|
2
|
+
connections, by reading config files only. No subprocess, no network.
|
|
3
|
+
|
|
4
|
+
Only connection NAMES and non-secret fields (authenticator, host, project)
|
|
5
|
+
ever leave this module. Tokens and passwords in the same files are dropped
|
|
6
|
+
at parse time so no caller can surface one by accident.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import configparser
|
|
12
|
+
import os
|
|
13
|
+
import re
|
|
14
|
+
import shutil
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
CLIS = {"snowflake": "snow", "bigquery": "bq", "databricks": "databricks"}
|
|
18
|
+
|
|
19
|
+
SAFE_CONNECTION_KEYS = ("authenticator", "account", "host")
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
def _parse_toml_tables(text: str) -> tuple[dict[str, dict], dict]:
|
|
23
|
+
"""(tables, top_level) for the flat TOML that snow writes.
|
|
24
|
+
|
|
25
|
+
tomllib on 3.11+; a line parser on 3.10. Either way, only string-ish
|
|
26
|
+
values survive and only SAFE_CONNECTION_KEYS are kept per table.
|
|
27
|
+
"""
|
|
28
|
+
try:
|
|
29
|
+
import tomllib
|
|
30
|
+
|
|
31
|
+
try:
|
|
32
|
+
data = tomllib.loads(text)
|
|
33
|
+
except tomllib.TOMLDecodeError:
|
|
34
|
+
return {}, {}
|
|
35
|
+
top = {k: v for k, v in data.items() if isinstance(v, str)}
|
|
36
|
+
tables = {}
|
|
37
|
+
for name, body in data.items():
|
|
38
|
+
if isinstance(body, dict):
|
|
39
|
+
tables[name] = body
|
|
40
|
+
return tables, top
|
|
41
|
+
except ImportError:
|
|
42
|
+
pass
|
|
43
|
+
tables: dict[str, dict] = {}
|
|
44
|
+
top: dict = {}
|
|
45
|
+
section = None
|
|
46
|
+
for line in text.splitlines():
|
|
47
|
+
line = line.strip()
|
|
48
|
+
if not line or line.startswith("#"):
|
|
49
|
+
continue
|
|
50
|
+
header = re.match(r"\[\s*([A-Za-z0-9_.-]+)\s*\]$", line)
|
|
51
|
+
if header:
|
|
52
|
+
section = header.group(1)
|
|
53
|
+
tables.setdefault(section, {})
|
|
54
|
+
continue
|
|
55
|
+
kv = re.match(r"([A-Za-z0-9_-]+)\s*=\s*(.+)$", line)
|
|
56
|
+
if not kv:
|
|
57
|
+
continue
|
|
58
|
+
value = kv.group(2).strip().strip("\"'")
|
|
59
|
+
if section is None:
|
|
60
|
+
top[kv.group(1)] = value
|
|
61
|
+
else:
|
|
62
|
+
tables[section][kv.group(1)] = value
|
|
63
|
+
return tables, top
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _connection_entry(body: dict) -> dict:
|
|
67
|
+
return {k: str(body[k]) for k in SAFE_CONNECTION_KEYS if isinstance(body.get(k), str)}
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def snowflake_connections() -> dict:
|
|
71
|
+
"""{"connections": {name: safe fields}, "default": name | None}.
|
|
72
|
+
|
|
73
|
+
connections.toml uses [name] sections; config.toml nests them as
|
|
74
|
+
[connections.name] and may set default_connection_name. Both files and
|
|
75
|
+
both shapes are read; SNOWFLAKE_HOME relocates the directory.
|
|
76
|
+
"""
|
|
77
|
+
home = Path(os.environ.get("SNOWFLAKE_HOME") or Path.home() / ".snowflake")
|
|
78
|
+
connections: dict[str, dict] = {}
|
|
79
|
+
default = os.environ.get("SNOWFLAKE_DEFAULT_CONNECTION_NAME")
|
|
80
|
+
for filename in ("connections.toml", "config.toml"):
|
|
81
|
+
try:
|
|
82
|
+
text = (home / filename).read_text(encoding="utf-8")
|
|
83
|
+
except OSError:
|
|
84
|
+
continue
|
|
85
|
+
tables, top = _parse_toml_tables(text)
|
|
86
|
+
if not default and isinstance(top.get("default_connection_name"), str):
|
|
87
|
+
default = top["default_connection_name"]
|
|
88
|
+
for name, body in tables.items():
|
|
89
|
+
if name == "connections":
|
|
90
|
+
# [connections] with nested tables (tomllib path)
|
|
91
|
+
for sub, sub_body in body.items():
|
|
92
|
+
if isinstance(sub_body, dict):
|
|
93
|
+
connections.setdefault(sub, _connection_entry(sub_body))
|
|
94
|
+
continue
|
|
95
|
+
if name.startswith("connections."):
|
|
96
|
+
name = name.split(".", 1)[1]
|
|
97
|
+
elif filename == "config.toml":
|
|
98
|
+
continue # config.toml top tables (cli, logs, ...) are not connections
|
|
99
|
+
connections.setdefault(name, _connection_entry(body))
|
|
100
|
+
return {"connections": connections, "default": default}
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def bigquery_configurations() -> dict:
|
|
104
|
+
"""{"configurations": {name: project | None}, "active": name | None}.
|
|
105
|
+
|
|
106
|
+
bq authenticates through gcloud, so a "connection" here is a gcloud
|
|
107
|
+
configuration and the project it points at.
|
|
108
|
+
"""
|
|
109
|
+
home = Path(os.environ.get("CLOUDSDK_CONFIG") or Path.home() / ".config" / "gcloud")
|
|
110
|
+
active = os.environ.get("CLOUDSDK_ACTIVE_CONFIG_NAME")
|
|
111
|
+
if not active:
|
|
112
|
+
try:
|
|
113
|
+
active = (home / "active_config").read_text(encoding="utf-8").strip() or None
|
|
114
|
+
except OSError:
|
|
115
|
+
active = None
|
|
116
|
+
configurations: dict[str, str | None] = {}
|
|
117
|
+
try:
|
|
118
|
+
config_files = sorted((home / "configurations").glob("config_*"))
|
|
119
|
+
except OSError:
|
|
120
|
+
config_files = []
|
|
121
|
+
for path in config_files:
|
|
122
|
+
name = path.name.removeprefix("config_")
|
|
123
|
+
parser = configparser.ConfigParser()
|
|
124
|
+
try:
|
|
125
|
+
parser.read_string(path.read_text(encoding="utf-8"))
|
|
126
|
+
except (OSError, configparser.Error):
|
|
127
|
+
continue
|
|
128
|
+
configurations[name] = parser.get("core", "project", fallback=None)
|
|
129
|
+
return {"configurations": configurations, "active": active}
|
|
130
|
+
|
|
131
|
+
|
|
132
|
+
def databricks_profiles() -> dict:
|
|
133
|
+
"""{name: {"host": ...}}. The token field in the same file is never read out."""
|
|
134
|
+
path = Path(os.environ.get("DATABRICKS_CONFIG_FILE") or Path.home() / ".databrickscfg")
|
|
135
|
+
# default_section="@" so [DEFAULT] shows up as an ordinary profile
|
|
136
|
+
parser = configparser.ConfigParser(default_section="@", interpolation=None)
|
|
137
|
+
try:
|
|
138
|
+
parser.read_string(path.read_text(encoding="utf-8"))
|
|
139
|
+
except (OSError, configparser.Error):
|
|
140
|
+
return {}
|
|
141
|
+
profiles: dict[str, dict] = {}
|
|
142
|
+
for name in parser.sections():
|
|
143
|
+
entry = {}
|
|
144
|
+
host = parser.get(name, "host", fallback=None)
|
|
145
|
+
if host:
|
|
146
|
+
entry["host"] = host
|
|
147
|
+
profiles[name] = entry
|
|
148
|
+
return profiles
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def capabilities() -> dict:
|
|
152
|
+
"""What is installed and configured, per platform. Names only, no secrets.
|
|
153
|
+
|
|
154
|
+
This is the first-call answer for collect: the caller shows it to the
|
|
155
|
+
user and asks which connection to use. Nothing here picks one.
|
|
156
|
+
"""
|
|
157
|
+
snowflake = snowflake_connections()
|
|
158
|
+
bigquery = bigquery_configurations()
|
|
159
|
+
databricks = databricks_profiles()
|
|
160
|
+
return {
|
|
161
|
+
"snowflake": {
|
|
162
|
+
"cli": "snow",
|
|
163
|
+
"cli_found": shutil.which("snow") is not None,
|
|
164
|
+
"connections": sorted(snowflake["connections"]),
|
|
165
|
+
"default": snowflake["default"],
|
|
166
|
+
"browser_sso": sorted(
|
|
167
|
+
name
|
|
168
|
+
for name, body in snowflake["connections"].items()
|
|
169
|
+
if body.get("authenticator", "").lower() == "externalbrowser"
|
|
170
|
+
),
|
|
171
|
+
},
|
|
172
|
+
"bigquery": {
|
|
173
|
+
"cli": "bq",
|
|
174
|
+
"cli_found": shutil.which("bq") is not None,
|
|
175
|
+
"configurations": [
|
|
176
|
+
{"name": name, "project": project}
|
|
177
|
+
for name, project in sorted(bigquery["configurations"].items())
|
|
178
|
+
],
|
|
179
|
+
"active": bigquery["active"],
|
|
180
|
+
},
|
|
181
|
+
"databricks": {
|
|
182
|
+
"cli": "databricks",
|
|
183
|
+
"cli_found": shutil.which("databricks") is not None,
|
|
184
|
+
"profiles": [{"name": name, **body} for name, body in sorted(databricks.items())],
|
|
185
|
+
},
|
|
186
|
+
"note": (
|
|
187
|
+
"Pick the platform and connection with the user; when several "
|
|
188
|
+
"connections exist, never choose one for them."
|
|
189
|
+
),
|
|
190
|
+
}
|