gen3-dataops-toolkit 2.0.0__tar.gz → 2.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/PKG-INFO +1 -1
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/pyproject.toml +1 -1
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/config_cmds.py +46 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/main.py +13 -3
- gen3_dataops_toolkit-2.1.0/src/g3dt/cli/pipeline_cmds.py +155 -0
- gen3_dataops_toolkit-2.1.0/src/g3dt/cli/release_cmds.py +74 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/utils/release_writer.py +99 -41
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/README.md +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/__init__.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/__init__.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/_internal/__init__.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/_internal/dispatch.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/_internal/registry.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/_internal/resolve.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/_internal/runner.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/_internal/safety.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/delete_cmds.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/dict_cmds.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/ec2_cmds.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/indexd_cmds.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/jobs.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/k8s.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/metadata.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/synth.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/config.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/indexd/__init__.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/indexd/indexd_registrar.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/ingest/ingest.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/resolver.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/delete/delete_all_metadata_for_project.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/delete/delete_metadata.sh +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/delete/delete_metadata_by_guid.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/dictionary/deploy_dd.sh +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/dictionary/pull_dict.sh +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/dictionary/upload_dictionary.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/indexd/register_indexd.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/k8s_ops/argocd_restart_etl.sh +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/k8s_ops/argocd_restart_ms.sh +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/k8s_ops/argocd_restart_schema.sh +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/k8s_ops/login_to_pod.sh +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/k8s_ops/restart_etl_and_ms.sh +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/synthetic_data/delete_synth_metadata_sheepdog.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/synthetic_data/full_deploy_dd_and_synth.sh +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/synthetic_data/generate_synth_metadata.sh +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/synthetic_data/upload_synth_metadata_sheepdog.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/upload/metadata/upload_all_studies.sh +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/upload/metadata/upload_metadata.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/upload/__init__.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/upload/metadata_deleter.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/upload/metadata_submitter.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/upload/upload_synthdata_s3.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/utils/athena_utils.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/utils/dbt_utils.py +0 -0
- {gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/validate/validate.py +0 -0
|
@@ -182,6 +182,52 @@ def diff(
|
|
|
182
182
|
raise typer.Exit(1 if drift else 0)
|
|
183
183
|
|
|
184
184
|
|
|
185
|
+
@app.command("dbt-env")
|
|
186
|
+
def dbt_env(
|
|
187
|
+
env: str = typer.Option(..., "--env", "-e", help="Environment, e.g. test."),
|
|
188
|
+
) -> None:
|
|
189
|
+
"""Emit `export` lines for the env's dbt settings (resolved from SSM).
|
|
190
|
+
|
|
191
|
+
The dbt template's profiles.yml / dbt_project.yml read their derived names
|
|
192
|
+
from env_var(); this command is the one source of those values, for both
|
|
193
|
+
CodeBuild and a laptop:
|
|
194
|
+
|
|
195
|
+
eval "$(g3dt config dbt-env --env test)" && dbt build
|
|
196
|
+
"""
|
|
197
|
+
import shlex
|
|
198
|
+
|
|
199
|
+
from g3dt import resolver
|
|
200
|
+
|
|
201
|
+
marker = config.load_marker()
|
|
202
|
+
project = config.require_project(marker)
|
|
203
|
+
base = config.env_base(env)
|
|
204
|
+
profile = None if env.endswith("_ec2") else config.aws_profile_for(base, marker)
|
|
205
|
+
try:
|
|
206
|
+
rc = resolver.resolve(project, base, profile=profile)
|
|
207
|
+
except config.ConfigError as exc:
|
|
208
|
+
typer.secho(str(exc), fg=typer.colors.RED, err=True)
|
|
209
|
+
raise typer.Exit(1)
|
|
210
|
+
|
|
211
|
+
values = {
|
|
212
|
+
"G3DT_REGION": rc.region,
|
|
213
|
+
"G3DT_ATHENA_WORKGROUP": rc.athena_workgroup,
|
|
214
|
+
"G3DT_ATHENA_OUTPUT": rc.athena_output_location,
|
|
215
|
+
"G3DT_DB_RAW_BRONZE": rc.get("glue/db/rawBronze"),
|
|
216
|
+
"G3DT_DB_RAW_SILVER": rc.get("glue/db/rawSilver"),
|
|
217
|
+
"G3DT_DB_RAW_GOLD": rc.get("glue/db/rawGold"),
|
|
218
|
+
"G3DT_S3_SILVER_DATA_DIR": f"s3://{rc.get('buckets/rawSilver')}/dbt/",
|
|
219
|
+
"G3DT_S3_GOLD_DATA_DIR": f"s3://{rc.get('buckets/rawGold')}/dbt/",
|
|
220
|
+
}
|
|
221
|
+
if profile:
|
|
222
|
+
# A named profile means a laptop run: select the dbt target that
|
|
223
|
+
# carries aws_profile_name (CodeBuild/EC2 stay on `default`, ambient).
|
|
224
|
+
values["G3DT_AWS_PROFILE"] = profile
|
|
225
|
+
values["G3DT_DBT_TARGET"] = "local"
|
|
226
|
+
for key, value in values.items():
|
|
227
|
+
if value is not None:
|
|
228
|
+
typer.echo(f"export {key}={shlex.quote(str(value))}")
|
|
229
|
+
|
|
230
|
+
|
|
185
231
|
@app.command("set")
|
|
186
232
|
def set_value(
|
|
187
233
|
key: str = typer.Argument(..., help="Bootstrap key: project, region, default_env."),
|
|
@@ -16,6 +16,8 @@ from g3dt.cli import (
|
|
|
16
16
|
jobs,
|
|
17
17
|
k8s,
|
|
18
18
|
metadata,
|
|
19
|
+
pipeline_cmds,
|
|
20
|
+
release_cmds,
|
|
19
21
|
synth,
|
|
20
22
|
)
|
|
21
23
|
|
|
@@ -34,6 +36,8 @@ app.add_typer(indexd_cmds.app, name="indexd")
|
|
|
34
36
|
app.add_typer(ec2_cmds.app, name="ec2")
|
|
35
37
|
app.add_typer(jobs.app, name="jobs")
|
|
36
38
|
app.add_typer(config_cmds.app, name="config")
|
|
39
|
+
app.add_typer(release_cmds.app, name="release")
|
|
40
|
+
app.add_typer(pipeline_cmds.app, name="pipeline")
|
|
37
41
|
|
|
38
42
|
|
|
39
43
|
_DOCS = """\
|
|
@@ -69,6 +73,12 @@ Typical release runbook (staging shown; repeat for prod with care)
|
|
|
69
73
|
3. g3dt jobs logs <run-id> --follow
|
|
70
74
|
4. g3dt k8s restart-etl --env staging
|
|
71
75
|
|
|
76
|
+
Data releases (the dbt pipeline; see the project's dbt repo)
|
|
77
|
+
git tag data-v1.4.0 && git push origin data-v1.4.0
|
|
78
|
+
g3dt pipeline status --env staging which stage is running/failed
|
|
79
|
+
g3dt pipeline logs --env staging --follow live dbt + release-writer output
|
|
80
|
+
(the pipeline itself runs `g3dt release write` — no names needed anywhere)
|
|
81
|
+
|
|
72
82
|
Synthetic data (test only, all local)
|
|
73
83
|
g3dt synth deploy --env test
|
|
74
84
|
|
|
@@ -78,9 +88,9 @@ EC2 / SSM prerequisites
|
|
|
78
88
|
- Local profile needs: ssm:SendCommand / ssm:GetCommandInvocation,
|
|
79
89
|
s3:GetObject on the log prefix, ec2:Start/Stop/DescribeInstances.
|
|
80
90
|
|
|
81
|
-
NOT run by this CLI: the Glue jobs (validation, release-JSON)
|
|
82
|
-
dbt pipelines
|
|
83
|
-
|
|
91
|
+
NOT run by this CLI: the Glue jobs (validation, release-JSON). The CodeBuild
|
|
92
|
+
dbt pipelines are triggered from the project's dbt repo (branch push = CI,
|
|
93
|
+
data-v* tag = release) and watched with `g3dt pipeline status|logs`.
|
|
84
94
|
"""
|
|
85
95
|
|
|
86
96
|
|
|
@@ -0,0 +1,155 @@
|
|
|
1
|
+
"""`g3dt pipeline` — watch the dbt CodePipeline / CodeBuild, mirroring `g3dt jobs`.
|
|
2
|
+
|
|
3
|
+
One way to watch any long thing: `g3dt jobs` tails EC2-dispatched runs;
|
|
4
|
+
`g3dt pipeline` does the same for the dbt pipelines. Pipeline and CodeBuild
|
|
5
|
+
project names resolve from the env's SSM tree, so the operator never types an
|
|
6
|
+
ARN or a console name.
|
|
7
|
+
"""
|
|
8
|
+
from __future__ import annotations
|
|
9
|
+
|
|
10
|
+
import time
|
|
11
|
+
from typing import Optional
|
|
12
|
+
|
|
13
|
+
import typer
|
|
14
|
+
|
|
15
|
+
from g3dt import config
|
|
16
|
+
from g3dt.upload.metadata_submitter import create_boto3_session
|
|
17
|
+
|
|
18
|
+
app = typer.Typer(no_args_is_help=True, help="Watch the dbt CodePipeline/CodeBuild.")
|
|
19
|
+
|
|
20
|
+
#: CodeBuild build statuses that mean the build is over.
|
|
21
|
+
_TERMINAL_BUILD_STATUSES = ("SUCCEEDED", "FAILED", "FAULT", "STOPPED", "TIMED_OUT")
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _resolved(env: str):
|
|
25
|
+
"""Return ``(rc, session)`` for the env (names from SSM, auth from marker)."""
|
|
26
|
+
from g3dt import resolver
|
|
27
|
+
|
|
28
|
+
marker = config.load_marker()
|
|
29
|
+
project = config.require_project(marker)
|
|
30
|
+
base = config.env_base(env)
|
|
31
|
+
profile = None if env.endswith("_ec2") else config.aws_profile_for(base, marker)
|
|
32
|
+
rc = resolver.resolve(project, base, profile=profile)
|
|
33
|
+
session = create_boto3_session(aws_profile=profile, aws_region=rc.region)
|
|
34
|
+
return rc, session
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
@app.command()
|
|
38
|
+
def status(
|
|
39
|
+
env: str = typer.Option(..., "--env", "-e", help="Environment, e.g. test."),
|
|
40
|
+
which: str = typer.Option(
|
|
41
|
+
"writeReleaseInfo",
|
|
42
|
+
"--which",
|
|
43
|
+
help="writeReleaseInfo | dbtTestAndRun",
|
|
44
|
+
),
|
|
45
|
+
) -> None:
|
|
46
|
+
"""Show the latest execution state per stage of the pipeline."""
|
|
47
|
+
try:
|
|
48
|
+
rc, session = _resolved(env)
|
|
49
|
+
except config.ConfigError as exc:
|
|
50
|
+
typer.secho(str(exc), fg=typer.colors.RED, err=True)
|
|
51
|
+
raise typer.Exit(1)
|
|
52
|
+
pipeline_name = rc.get(f"codepipeline/{which}")
|
|
53
|
+
if not pipeline_name:
|
|
54
|
+
typer.secho(
|
|
55
|
+
f"No SSM parameter codepipeline/{which} — valid values: "
|
|
56
|
+
f"writeReleaseInfo, dbtTestAndRun.",
|
|
57
|
+
fg=typer.colors.RED,
|
|
58
|
+
err=True,
|
|
59
|
+
)
|
|
60
|
+
raise typer.Exit(2)
|
|
61
|
+
state = session.client("codepipeline").get_pipeline_state(name=pipeline_name)
|
|
62
|
+
typer.secho(f"{pipeline_name}", bold=True)
|
|
63
|
+
for stage in state.get("stageStates", []):
|
|
64
|
+
latest = stage.get("latestExecution", {})
|
|
65
|
+
typer.echo(f" {stage['stageName']:<32} {latest.get('status', 'n/a')}")
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@app.command()
|
|
69
|
+
def logs(
|
|
70
|
+
env: str = typer.Option(..., "--env", "-e", help="Environment, e.g. test."),
|
|
71
|
+
which: str = typer.Option(
|
|
72
|
+
"dbtReleaseBuilder",
|
|
73
|
+
"--which",
|
|
74
|
+
help="dbtReleaseBuilder | dbtTestAndRun",
|
|
75
|
+
),
|
|
76
|
+
follow: bool = typer.Option(False, "--follow", "-f", help="Tail until the build ends."),
|
|
77
|
+
poll_seconds: int = typer.Option(5, "--poll-seconds", help="Follow poll interval."),
|
|
78
|
+
) -> None:
|
|
79
|
+
"""Print (and optionally follow) the latest build's CloudWatch output.
|
|
80
|
+
|
|
81
|
+
The log group is CodeBuild's default ``/aws/codebuild/<project>``; the
|
|
82
|
+
project name comes from SSM ``codebuild/*``. The follow loop is the same
|
|
83
|
+
filter_log_events + de-dup pattern as `g3dt jobs logs`.
|
|
84
|
+
"""
|
|
85
|
+
try:
|
|
86
|
+
rc, session = _resolved(env)
|
|
87
|
+
except config.ConfigError as exc:
|
|
88
|
+
typer.secho(str(exc), fg=typer.colors.RED, err=True)
|
|
89
|
+
raise typer.Exit(1)
|
|
90
|
+
project_name = rc.get(f"codebuild/{which}")
|
|
91
|
+
if not project_name:
|
|
92
|
+
typer.secho(
|
|
93
|
+
f"No SSM parameter codebuild/{which} — valid values: "
|
|
94
|
+
f"dbtReleaseBuilder, dbtTestAndRun.",
|
|
95
|
+
fg=typer.colors.RED,
|
|
96
|
+
err=True,
|
|
97
|
+
)
|
|
98
|
+
raise typer.Exit(2)
|
|
99
|
+
|
|
100
|
+
cb = session.client("codebuild")
|
|
101
|
+
build_ids = cb.list_builds_for_project(
|
|
102
|
+
projectName=project_name, sortOrder="DESCENDING"
|
|
103
|
+
).get("ids", [])
|
|
104
|
+
if not build_ids:
|
|
105
|
+
typer.secho(f"No builds yet for {project_name}.", fg=typer.colors.YELLOW)
|
|
106
|
+
return
|
|
107
|
+
build_id = build_ids[0]
|
|
108
|
+
# CodeBuild names the stream <project>/<build-uuid> in /aws/codebuild/<project>.
|
|
109
|
+
stream_prefix = build_id.split(":", 1)[1] if ":" in build_id else build_id
|
|
110
|
+
group = f"/aws/codebuild/{project_name}"
|
|
111
|
+
logs_client = session.client("logs")
|
|
112
|
+
|
|
113
|
+
def build_status() -> Optional[str]:
|
|
114
|
+
builds = cb.batch_get_builds(ids=[build_id]).get("builds", [])
|
|
115
|
+
return builds[0].get("buildStatus") if builds else None
|
|
116
|
+
|
|
117
|
+
typer.secho(f"{build_id} ({group})", bold=True)
|
|
118
|
+
seen_ids: set = set()
|
|
119
|
+
last_ts = 0
|
|
120
|
+
|
|
121
|
+
def drain() -> None:
|
|
122
|
+
nonlocal last_ts
|
|
123
|
+
kwargs = {
|
|
124
|
+
"logGroupName": group,
|
|
125
|
+
"logStreamNamePrefix": stream_prefix,
|
|
126
|
+
"startTime": last_ts,
|
|
127
|
+
}
|
|
128
|
+
events = []
|
|
129
|
+
try:
|
|
130
|
+
while True:
|
|
131
|
+
resp = logs_client.filter_log_events(**kwargs)
|
|
132
|
+
events.extend(resp.get("events", []))
|
|
133
|
+
token = resp.get("nextToken")
|
|
134
|
+
if not token:
|
|
135
|
+
break
|
|
136
|
+
kwargs["nextToken"] = token
|
|
137
|
+
except logs_client.exceptions.ResourceNotFoundException:
|
|
138
|
+
return
|
|
139
|
+
for ev in sorted(events, key=lambda e: e["timestamp"]):
|
|
140
|
+
if ev["eventId"] in seen_ids:
|
|
141
|
+
continue
|
|
142
|
+
seen_ids.add(ev["eventId"])
|
|
143
|
+
last_ts = max(last_ts, ev["timestamp"])
|
|
144
|
+
typer.echo(ev["message"].rstrip("\n"))
|
|
145
|
+
|
|
146
|
+
while True:
|
|
147
|
+
drain()
|
|
148
|
+
if not follow:
|
|
149
|
+
break
|
|
150
|
+
current = build_status()
|
|
151
|
+
if current in _TERMINAL_BUILD_STATUSES:
|
|
152
|
+
drain() # catch events ingested between the last drain and completion
|
|
153
|
+
typer.secho(f"\n[build finished: {current}]", fg=typer.colors.BRIGHT_BLACK)
|
|
154
|
+
break
|
|
155
|
+
time.sleep(poll_seconds)
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
"""`g3dt release` — write the dbt data-release manifest to Athena.
|
|
2
|
+
|
|
3
|
+
The data half of the release decoupling: a `data-v*` tag on a project's dbt
|
|
4
|
+
repo drives CodePipeline → CodeBuild → `g3dt release write`, which records one
|
|
5
|
+
idempotent row per dbt model in the env's `releases` Iceberg table. Every name
|
|
6
|
+
it needs — release DB/table, the metadata bucket the table lives under, the
|
|
7
|
+
Athena workgroup output, the region — resolves from the env's SSM tree, so the
|
|
8
|
+
buildspec passes only `--env`, the version, and the commit SHA.
|
|
9
|
+
"""
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import typer
|
|
13
|
+
|
|
14
|
+
from g3dt import config
|
|
15
|
+
from g3dt.utils import release_writer
|
|
16
|
+
|
|
17
|
+
app = typer.Typer(no_args_is_help=True, help="Write/inspect dbt data releases.")
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@app.command()
|
|
21
|
+
def write(
|
|
22
|
+
env: str = typer.Option(..., "--env", "-e", help="Environment, e.g. test."),
|
|
23
|
+
data_release_version: str = typer.Option(
|
|
24
|
+
...,
|
|
25
|
+
"--data-release-version",
|
|
26
|
+
help="Data version, e.g. 1.4.0 (from the data-v* tag, prefix stripped).",
|
|
27
|
+
),
|
|
28
|
+
commit_id: str = typer.Option("", "--commit-id", help="Git SHA for auditing."),
|
|
29
|
+
dbt_schema_path: str = typer.Option(
|
|
30
|
+
"models/schema.yml",
|
|
31
|
+
"--dbt-schema-path",
|
|
32
|
+
help="dbt schema file listing the models to track (relative to the dbt project root).",
|
|
33
|
+
),
|
|
34
|
+
dry_run: bool = typer.Option(
|
|
35
|
+
False, "--dry-run", help="Print the resolved DB/table/SQL and write nothing."
|
|
36
|
+
),
|
|
37
|
+
) -> None:
|
|
38
|
+
"""Resolve release/db, release/table, athena/* from SSM, then write the rows.
|
|
39
|
+
|
|
40
|
+
Idempotent: a (release_tag, model, db) row that already exists is skipped,
|
|
41
|
+
so re-running the same tag is safe.
|
|
42
|
+
"""
|
|
43
|
+
from g3dt import resolver
|
|
44
|
+
|
|
45
|
+
marker = config.load_marker()
|
|
46
|
+
project = config.require_project(marker)
|
|
47
|
+
base = config.env_base(env)
|
|
48
|
+
profile = None if env.endswith("_ec2") else config.aws_profile_for(base, marker)
|
|
49
|
+
|
|
50
|
+
try:
|
|
51
|
+
rc = resolver.resolve(project, base, profile=profile)
|
|
52
|
+
typer.secho(
|
|
53
|
+
f"Release target (from SSM /{project}/{base}): "
|
|
54
|
+
f"{rc.release_db}.{rc.release_table} "
|
|
55
|
+
f"(workgroup output {rc.athena_output_location})",
|
|
56
|
+
bold=True,
|
|
57
|
+
)
|
|
58
|
+
release_writer.run(
|
|
59
|
+
dbt_schema_path=dbt_schema_path,
|
|
60
|
+
release_db=rc.release_db,
|
|
61
|
+
release_table=rc.release_table,
|
|
62
|
+
release_s3_location=f"s3://{rc.metadata_bucket}/",
|
|
63
|
+
data_release_version=data_release_version,
|
|
64
|
+
commit_id=commit_id,
|
|
65
|
+
aws_region=rc.region,
|
|
66
|
+
athena_s3_output=rc.athena_output_location,
|
|
67
|
+
aws_profile=profile,
|
|
68
|
+
dry_run=dry_run,
|
|
69
|
+
)
|
|
70
|
+
except config.ConfigError as exc:
|
|
71
|
+
typer.secho(str(exc), fg=typer.colors.RED, err=True)
|
|
72
|
+
raise typer.Exit(1)
|
|
73
|
+
if dry_run:
|
|
74
|
+
typer.secho("Dry run complete — nothing was written.", fg=typer.colors.GREEN)
|
|
@@ -38,7 +38,8 @@ def insert_release_row(
|
|
|
38
38
|
release_db: str,
|
|
39
39
|
release_table: str,
|
|
40
40
|
release_tag: str,
|
|
41
|
-
github_sha: str
|
|
41
|
+
github_sha: str,
|
|
42
|
+
dry_run: bool = False,
|
|
42
43
|
) -> None:
|
|
43
44
|
"""
|
|
44
45
|
Insert a new row into the release tracking table for a given model and snapshot.
|
|
@@ -46,35 +47,40 @@ def insert_release_row(
|
|
|
46
47
|
This function checks if a row with the same release_tag, model_name, and db_name already exists
|
|
47
48
|
in the release table to prevent duplicate entries. If no such row exists, it inserts a new row
|
|
48
49
|
with the provided snapshot_id, committed_at timestamp, and other metadata.
|
|
50
|
+
|
|
51
|
+
With ``dry_run`` the INSERT is built and logged but never executed (and the
|
|
52
|
+
existence check is skipped), so an operator can confirm the exact target
|
|
53
|
+
table and SQL before a real release.
|
|
49
54
|
"""
|
|
50
55
|
athena_query = AthenaQuery(athena_config)
|
|
51
56
|
if not db_name:
|
|
52
57
|
logger.warning(f"Skipping model '{model_name}': No database found for this model.")
|
|
53
58
|
return
|
|
54
59
|
|
|
55
|
-
|
|
56
|
-
|
|
57
|
-
|
|
58
|
-
check_sql = f"""
|
|
59
|
-
SELECT COUNT(*) AS cnt
|
|
60
|
-
FROM "{release_db}"."{release_table}"
|
|
61
|
-
WHERE release_tag = {safe_sql_string(release_tag)}
|
|
62
|
-
AND model_name = {safe_sql_string(model_name)}
|
|
63
|
-
AND db_name = {safe_sql_string(db_name)}
|
|
64
|
-
"""
|
|
65
|
-
try:
|
|
66
|
-
cnt_df = athena_query.query_athena(check_sql, release_db)
|
|
67
|
-
cnt = cnt_df.iloc[0]['cnt'] if not cnt_df.empty else 0
|
|
68
|
-
except Exception as e:
|
|
69
|
-
logger.error(
|
|
70
|
-
f"Could not query existing release rows for release_tag='{release_tag}', model='{model_name}', db='{db_name}': {e}",
|
|
71
|
-
exc_info=True
|
|
60
|
+
if not dry_run:
|
|
61
|
+
logger.debug(
|
|
62
|
+
f"Checking for existing release row: release_tag={release_tag!r}, model_name={model_name!r}, db_name={db_name!r}"
|
|
72
63
|
)
|
|
73
|
-
|
|
74
|
-
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
|
|
64
|
+
check_sql = f"""
|
|
65
|
+
SELECT COUNT(*) AS cnt
|
|
66
|
+
FROM "{release_db}"."{release_table}"
|
|
67
|
+
WHERE release_tag = {safe_sql_string(release_tag)}
|
|
68
|
+
AND model_name = {safe_sql_string(model_name)}
|
|
69
|
+
AND db_name = {safe_sql_string(db_name)}
|
|
70
|
+
"""
|
|
71
|
+
try:
|
|
72
|
+
cnt_df = athena_query.query_athena(check_sql, release_db)
|
|
73
|
+
cnt = cnt_df.iloc[0]['cnt'] if not cnt_df.empty else 0
|
|
74
|
+
except Exception as e:
|
|
75
|
+
logger.error(
|
|
76
|
+
f"Could not query existing release rows for release_tag='{release_tag}', model='{model_name}', db='{db_name}': {e}",
|
|
77
|
+
exc_info=True
|
|
78
|
+
)
|
|
79
|
+
raise
|
|
80
|
+
|
|
81
|
+
if cnt > 0:
|
|
82
|
+
logger.info(f"[SKIP] Release row already exists: {release_tag} / {db_name}.{model_name}")
|
|
83
|
+
return
|
|
78
84
|
|
|
79
85
|
snap_val = "NULL" if snapshot_id is None else str(snapshot_id)
|
|
80
86
|
commit_val = f"TIMESTAMP '{committed_at}'" if committed_at else "NULL"
|
|
@@ -93,6 +99,13 @@ def insert_release_row(
|
|
|
93
99
|
{sha_val}
|
|
94
100
|
)
|
|
95
101
|
"""
|
|
102
|
+
if dry_run:
|
|
103
|
+
logger.info(
|
|
104
|
+
f"[DRY-RUN] would insert into {release_db}.{release_table} for "
|
|
105
|
+
f"{db_name}.{model_name} [release_tag={release_tag}]:\n{insert_sql}"
|
|
106
|
+
)
|
|
107
|
+
return
|
|
108
|
+
|
|
96
109
|
logger.info(
|
|
97
110
|
f"Inserting new release row for {db_name}.{model_name} [release_tag={release_tag}, snapshot_id={snapshot_id}, committed_at={committed_at}]"
|
|
98
111
|
)
|
|
@@ -130,31 +143,55 @@ def parse_args():
|
|
|
130
143
|
help="S3 URI for Athena query results (e.g., 's3://athena-results-bucket/').")
|
|
131
144
|
parser.add_argument("--release-s3-location", type=str, required=True,
|
|
132
145
|
help="S3 location for the release table (e.g., 's3://<metadata-bucket>/').")
|
|
146
|
+
parser.add_argument("--dry-run", action="store_true", default=False,
|
|
147
|
+
help="Resolve everything and log the SQL; write nothing.")
|
|
133
148
|
parser.add_argument("-v", "--verbose", action="store_true", default=False,
|
|
134
149
|
help="Enable debug logging.")
|
|
135
150
|
return parser.parse_args()
|
|
136
151
|
|
|
137
|
-
def main():
|
|
138
|
-
args = parse_args()
|
|
139
|
-
log_level = logging.DEBUG if args.verbose else logging.INFO
|
|
140
|
-
logging.getLogger().setLevel(log_level)
|
|
141
152
|
|
|
153
|
+
def run(
|
|
154
|
+
*,
|
|
155
|
+
dbt_schema_path: str,
|
|
156
|
+
release_db: str,
|
|
157
|
+
release_table: str,
|
|
158
|
+
release_s3_location: str,
|
|
159
|
+
data_release_version: str,
|
|
160
|
+
commit_id: str,
|
|
161
|
+
aws_region: str,
|
|
162
|
+
athena_s3_output: str,
|
|
163
|
+
aws_profile: Optional[str] = None,
|
|
164
|
+
dry_run: bool = False,
|
|
165
|
+
) -> None:
|
|
166
|
+
"""Write one release row per dbt model (idempotent; see insert_release_row).
|
|
167
|
+
|
|
168
|
+
The callable core behind both entry points: the argv-driven ``main()``
|
|
169
|
+
(``python -m g3dt.utils.release_writer``) and the SSM-resolving CLI wrapper
|
|
170
|
+
(``g3dt release write``). With ``dry_run`` the resolved target and the
|
|
171
|
+
INSERT SQL are logged but nothing is written (model/snapshot lookups are
|
|
172
|
+
read-only and still run).
|
|
173
|
+
"""
|
|
142
174
|
logger.info("==== DBT release snapshot info writer started ====")
|
|
143
|
-
logger.debug(f"Parsed arguments: {args}")
|
|
144
175
|
|
|
145
176
|
athena_config = AthenaConfig(
|
|
146
|
-
aws_region=
|
|
147
|
-
aws_profile=
|
|
148
|
-
athena_s3_output=
|
|
177
|
+
aws_region=aws_region,
|
|
178
|
+
aws_profile=aws_profile,
|
|
179
|
+
athena_s3_output=athena_s3_output
|
|
149
180
|
)
|
|
150
181
|
|
|
151
|
-
dbt_models = get_model_names(
|
|
182
|
+
dbt_models = get_model_names(dbt_schema_path)
|
|
152
183
|
athena_query = AthenaQuery(athena_config)
|
|
153
|
-
|
|
154
|
-
|
|
155
|
-
|
|
156
|
-
|
|
157
|
-
|
|
184
|
+
|
|
185
|
+
if dry_run:
|
|
186
|
+
logger.info(
|
|
187
|
+
f"[DRY-RUN] would ensure release table {release_db}.{release_table} "
|
|
188
|
+
f"at {release_s3_location}"
|
|
189
|
+
)
|
|
190
|
+
else:
|
|
191
|
+
logger.info(f"Creating release table: {release_db}.{release_table}")
|
|
192
|
+
athena_query.create_release_table(
|
|
193
|
+
release_db, release_table, release_s3_location
|
|
194
|
+
)
|
|
158
195
|
|
|
159
196
|
logger.info(f"Processing DBT models from schema: {dbt_models}")
|
|
160
197
|
|
|
@@ -175,14 +212,35 @@ def main():
|
|
|
175
212
|
db_name=db_name,
|
|
176
213
|
snapshot_id=snapshot_id,
|
|
177
214
|
committed_at=commit_datetime,
|
|
178
|
-
release_db=
|
|
179
|
-
release_table=
|
|
180
|
-
release_tag=
|
|
181
|
-
github_sha=
|
|
215
|
+
release_db=release_db,
|
|
216
|
+
release_table=release_table,
|
|
217
|
+
release_tag=data_release_version,
|
|
218
|
+
github_sha=commit_id,
|
|
219
|
+
dry_run=dry_run,
|
|
182
220
|
)
|
|
183
221
|
logger.info(f"[SUCCESS] Release info recorded for {db_name}.{model_name}")
|
|
184
222
|
|
|
185
223
|
logger.info(f"Finished release process for DBT models: {dbt_models}")
|
|
186
224
|
|
|
225
|
+
|
|
226
|
+
def main():
|
|
227
|
+
args = parse_args()
|
|
228
|
+
log_level = logging.DEBUG if args.verbose else logging.INFO
|
|
229
|
+
logging.getLogger().setLevel(log_level)
|
|
230
|
+
logger.debug(f"Parsed arguments: {args}")
|
|
231
|
+
|
|
232
|
+
run(
|
|
233
|
+
dbt_schema_path=args.dbt_schema_path,
|
|
234
|
+
release_db=args.release_db,
|
|
235
|
+
release_table=args.release_table,
|
|
236
|
+
release_s3_location=args.release_s3_location,
|
|
237
|
+
data_release_version=args.data_release_version,
|
|
238
|
+
commit_id=args.commit_id,
|
|
239
|
+
aws_region=args.aws_region,
|
|
240
|
+
athena_s3_output=args.athena_s3_output,
|
|
241
|
+
aws_profile=args.aws_profile,
|
|
242
|
+
dry_run=args.dry_run,
|
|
243
|
+
)
|
|
244
|
+
|
|
187
245
|
if __name__ == "__main__":
|
|
188
246
|
main()
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/_internal/__init__.py
RENAMED
|
File without changes
|
{gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/_internal/dispatch.py
RENAMED
|
File without changes
|
{gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/cli/_internal/registry.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/indexd/indexd_registrar.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/dictionary/deploy_dd.sh
RENAMED
|
File without changes
|
{gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/dictionary/pull_dict.sh
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/services/k8s_ops/login_to_pod.sh
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/upload/metadata_deleter.py
RENAMED
|
File without changes
|
{gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/upload/metadata_submitter.py
RENAMED
|
File without changes
|
{gen3_dataops_toolkit-2.0.0 → gen3_dataops_toolkit-2.1.0}/src/g3dt/upload/upload_synthdata_s3.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|