gen3-dataops-toolkit 4.0.1__tar.gz → 4.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (62) hide show
  1. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/PKG-INFO +7 -3
  2. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/README.md +6 -2
  3. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/pyproject.toml +2 -2
  4. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/_internal/resolve.py +41 -0
  5. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/config_cmds.py +4 -28
  6. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/delete_cmds.py +83 -11
  7. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/main.py +15 -6
  8. gen3_dataops_toolkit-4.2.0/src/g3dt/cli/study_cmds.py +580 -0
  9. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/synth.py +9 -0
  10. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/config.py +134 -49
  11. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/resolver.py +3 -0
  12. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/delete/delete_all_metadata_for_project.py +25 -5
  13. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/delete/delete_metadata.sh +29 -7
  14. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/delete/delete_metadata_by_guid.py +1 -1
  15. gen3_dataops_toolkit-4.2.0/src/g3dt/services/delete/delete_synth_metadata_by_version.py +334 -0
  16. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/dictionary/upload_dictionary.py +13 -8
  17. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/indexd/register_indexd.py +2 -2
  18. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/synthetic_data/generate_synth_metadata.sh +8 -0
  19. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/upload/metadata/upload_metadata.py +1 -1
  20. gen3_dataops_toolkit-4.2.0/src/g3dt/studies.py +484 -0
  21. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/upload/metadata_deleter.py +78 -0
  22. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/__init__.py +0 -0
  23. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/__init__.py +0 -0
  24. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/_internal/__init__.py +0 -0
  25. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/_internal/aws_quiet.py +0 -0
  26. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/_internal/dispatch.py +0 -0
  27. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/_internal/helptext.py +0 -0
  28. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/_internal/registry.py +0 -0
  29. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/_internal/runner.py +0 -0
  30. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/_internal/safety.py +0 -0
  31. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/dict_cmds.py +0 -0
  32. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/ec2_cmds.py +0 -0
  33. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/indexd_cmds.py +0 -0
  34. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/jobs.py +0 -0
  35. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/k8s.py +0 -0
  36. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/metadata.py +0 -0
  37. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/pipeline_cmds.py +0 -0
  38. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/cli/release_cmds.py +0 -0
  39. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/contexts.py +0 -0
  40. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/indexd/__init__.py +0 -0
  41. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/indexd/file_access.py +0 -0
  42. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/indexd/indexd_registrar.py +0 -0
  43. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/ingest/ingest.py +0 -0
  44. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/dictionary/deploy_dd.sh +0 -0
  45. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/dictionary/pull_dict.sh +0 -0
  46. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/indexd/verify_file_access.py +0 -0
  47. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/k8s_ops/argocd_restart_etl.sh +0 -0
  48. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/k8s_ops/argocd_restart_ms.sh +0 -0
  49. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/k8s_ops/argocd_restart_schema.sh +0 -0
  50. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/k8s_ops/login_to_pod.sh +0 -0
  51. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/k8s_ops/restart_etl_and_ms.sh +0 -0
  52. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/synthetic_data/delete_synth_metadata_sheepdog.py +0 -0
  53. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/synthetic_data/full_deploy_dd_and_synth.sh +0 -0
  54. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/synthetic_data/upload_synth_metadata_sheepdog.py +0 -0
  55. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/services/upload/metadata/upload_all_studies.sh +0 -0
  56. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/upload/__init__.py +0 -0
  57. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/upload/metadata_submitter.py +0 -0
  58. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/upload/upload_synthdata_s3.py +0 -0
  59. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/utils/athena_utils.py +0 -0
  60. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/utils/dbt_utils.py +0 -0
  61. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/utils/release_writer.py +0 -0
  62. {gen3_dataops_toolkit-4.0.1 → gen3_dataops_toolkit-4.2.0}/src/g3dt/validate/validate.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gen3-dataops-toolkit
3
- Version: 4.0.1
3
+ Version: 4.2.0
4
4
  Summary: Gen3 DataOps toolkit (g3dt): operate SSM-published Gen3 data pipeline environments
5
5
  License: Apache-2.0
6
6
  Author: JoshuaHarris391
@@ -80,8 +80,11 @@ confirmation, and destructive actions on them require typing the context
80
80
  name — `--yes` never bypasses that.
81
81
 
82
82
  The marker can also be written by hand (design doc:
83
- `docs/design/contexts.md`); the study registry lives either in a top-level
84
- `studies:` block or per env at `s3://<metadata-bucket>/config/studies.yaml`.
83
+ `docs/design/contexts.md`). The study registry lives in the env's SSM tree
84
+ (`/{project}/{env}/studies/<name>`, one JSON parameter per study — manage
85
+ it with `g3dt study`; design doc `docs/design/studies.md`). The legacy
86
+ per-env `s3://<metadata-bucket>/config/studies.yaml` is a deprecation-
87
+ warned read fallback until 5.0 — import it with `g3dt study migrate`.
85
88
  Search order: `./g3dt.yaml` → `~/.g3dt/g3dt.yaml` → `/etc/g3dt/g3dt.yaml`
86
89
  (the EC2 job box's copy, written by CDK user-data). Legacy markers
87
90
  (`project`/`default_env`/`profiles:` keys) keep working unchanged, as do the
@@ -92,6 +95,7 @@ the file-less CodeBuild/EC2 path.
92
95
 
93
96
  ```bash
94
97
  g3dt config show # every resolved name — the safety check
98
+ g3dt study repoint --latest # point the registry at the newest release
95
99
  g3dt ec2 up # start the context's job box (SSM-managed)
96
100
  g3dt metadata upload --study mystudy --on ec2
97
101
  g3dt jobs logs <run-id> --follow # live logs; laptop can sleep, job keeps going
@@ -46,8 +46,11 @@ confirmation, and destructive actions on them require typing the context
46
46
  name — `--yes` never bypasses that.
47
47
 
48
48
  The marker can also be written by hand (design doc:
49
- `docs/design/contexts.md`); the study registry lives either in a top-level
50
- `studies:` block or per env at `s3://<metadata-bucket>/config/studies.yaml`.
49
+ `docs/design/contexts.md`). The study registry lives in the env's SSM tree
50
+ (`/{project}/{env}/studies/<name>`, one JSON parameter per study — manage
51
+ it with `g3dt study`; design doc `docs/design/studies.md`). The legacy
52
+ per-env `s3://<metadata-bucket>/config/studies.yaml` is a deprecation-
53
+ warned read fallback until 5.0 — import it with `g3dt study migrate`.
51
54
  Search order: `./g3dt.yaml` → `~/.g3dt/g3dt.yaml` → `/etc/g3dt/g3dt.yaml`
52
55
  (the EC2 job box's copy, written by CDK user-data). Legacy markers
53
56
  (`project`/`default_env`/`profiles:` keys) keep working unchanged, as do the
@@ -58,6 +61,7 @@ the file-less CodeBuild/EC2 path.
58
61
 
59
62
  ```bash
60
63
  g3dt config show # every resolved name — the safety check
64
+ g3dt study repoint --latest # point the registry at the newest release
61
65
  g3dt ec2 up # start the context's job box (SSM-managed)
62
66
  g3dt metadata upload --study mystudy --on ec2
63
67
  g3dt jobs logs <run-id> --follow # live logs; laptop can sleep, job keeps going
@@ -1,6 +1,6 @@
1
1
  [tool.poetry]
2
2
  name = "gen3-dataops-toolkit"
3
- version = "4.0.1"
3
+ version = "4.2.0"
4
4
  description = "Gen3 DataOps toolkit (g3dt): operate SSM-published Gen3 data pipeline environments"
5
5
  authors = ["JoshuaHarris391 <harjo391@gmail.com>"]
6
6
  readme = "README.md"
@@ -44,7 +44,7 @@ moto = ">=5.0"
44
44
  optional = true
45
45
 
46
46
  [tool.poetry.group.synth.dependencies]
47
- gen3-metadata-simulator = "^0.3.0"
47
+ gen3-metadata-simulator = "^0.5.3"
48
48
 
49
49
  [build-system]
50
50
  requires = ["poetry-core>=2.0.0,<3.0.0"]
@@ -30,6 +30,28 @@ def study_of(study: str, env: str) -> config.StudyConfig:
30
30
  except config.ConfigError as exc:
31
31
  typer.secho(str(exc), fg=typer.colors.RED, err=True)
32
32
  raise typer.Exit(1)
33
+ except Exception as exc: # botocore auth/permission failures, guided
34
+ typer.secho(_aws_error_message(exc, env), fg=typer.colors.RED, err=True)
35
+ raise typer.Exit(1)
36
+
37
+
38
+ def _aws_error_message(exc: Exception, env: str) -> str:
39
+ """A guided rendering of a raw AWS/botocore failure.
40
+
41
+ An expired SSO session used to surface as either a traceback or —
42
+ worse — a silent "(none)" study registry. Name the likely fix instead.
43
+ """
44
+ profile = None
45
+ try:
46
+ profile = config.aws_profile_for(env, config.load_marker())
47
+ except Exception:
48
+ pass
49
+ hint = (
50
+ f" If your SSO session expired: aws sso login --profile {profile}"
51
+ if profile
52
+ else ""
53
+ )
54
+ return f"AWS error while resolving env '{env}': {exc}.{hint}"
33
55
 
34
56
 
35
57
  def active_env(env: Optional[str]) -> str:
@@ -83,3 +105,22 @@ def rc_of(env: str):
83
105
  except config.ConfigError as exc:
84
106
  typer.secho(str(exc), fg=typer.colors.RED, err=True)
85
107
  raise typer.Exit(1)
108
+ except Exception as exc: # botocore auth/permission failures, guided
109
+ typer.secho(_aws_error_message(exc, env), fg=typer.colors.RED, err=True)
110
+ raise typer.Exit(1)
111
+
112
+
113
+ def rc_session_of(env: str):
114
+ """``(rc, session)`` for ``env`` — resolved names plus an authed session.
115
+
116
+ The single idiom for commands that both read SSM names and make further
117
+ AWS calls (Athena, S3, SSM writes). Same credential rule as :func:`rc_of`;
118
+ the session's region is the resolved env's own ``meta/region``.
119
+ """
120
+ import boto3
121
+
122
+ rc = rc_of(env)
123
+ marker = config.load_marker()
124
+ profile = None if env.endswith("_ec2") else config.aws_profile_for(env, marker)
125
+ session = boto3.Session(profile_name=profile, region_name=rc.region)
126
+ return rc, session
@@ -528,36 +528,12 @@ def envs() -> None:
528
528
 
529
529
  @app.command()
530
530
  def studies(
531
- env: Optional[str] = typer.Option(
532
- None,
533
- "--env",
534
- "-e",
535
- help="Read the study registry from this env's S3 bucket "
536
- "(s3://<metadata-bucket>/config/studies.yaml) instead of the "
537
- "local marker — this changes the data source, not just the "
538
- "target.",
539
- ),
531
+ env: Optional[str] = typer.Option(None, "--env", "-e", help=ENV_OPT),
540
532
  ) -> None:
541
- """List the configured studies (bare names).
533
+ """List the env's registered studies (alias of `g3dt study list`)."""
534
+ from g3dt.cli import study_cmds
542
535
 
543
- The registry comes from the marker's studies: block, or — pass --env —
544
- from the env's S3 registry, which is what the EC2 job box uses.
545
- """
546
- if env is not None:
547
- env = resolve.active_env(env)
548
- else:
549
- resolve.announce_context()
550
- names = config.list_studies(env=env)
551
- if not names:
552
- typer.secho(
553
- "No studies configured. Add a studies: block to your g3dt.yaml "
554
- "marker, or upload config/studies.yaml to the env's metadata "
555
- "bucket (and pass --env).",
556
- fg=typer.colors.YELLOW,
557
- )
558
- return
559
- for name in names:
560
- typer.echo(name)
536
+ study_cmds.list_impl(env)
561
537
 
562
538
 
563
539
  @app.command()
@@ -7,6 +7,13 @@ the bare names (a specific version like ``0.9.8``, resolved via an Athena GUID
7
7
  lookup, or ``all`` for every version). A bare study with no version anywhere
8
8
  is refused (exit 2).
9
9
 
10
+ ``--synthetic`` switches to registry-free mode for synthetic data: each
11
+ ``--studies`` name is the Gen3 project code itself (no SSM study registry —
12
+ synthetic projects are never registered), bare names default to version
13
+ ``all``, and a specific version is matched verbatim against the records'
14
+ ``data_version`` property via GraphQL rather than Athena receipts (synthetic
15
+ uploads write none).
16
+
10
17
  Every command confirms before acting. Production always requires typing the
11
18
  target id, even with ``--yes``. Deleting ALL versions always prompts, even with
12
19
  ``--yes``. Confirmation happens locally before any EC2 dispatch (SSM has no
@@ -61,12 +68,30 @@ def _normalise_version(raw: str, where: str) -> str:
61
68
  return match.group(1)
62
69
 
63
70
 
64
- def _parse_study_specs(studies: str, fallback, env: str):
71
+ def _synthetic_version(raw: str) -> str:
72
+ """Canonicalise a synthetic version token: ``all`` (any case) or verbatim.
73
+
74
+ Synthetic versions are matched exactly against the records' ``data_version``
75
+ property, and the natural label there is the dictionary version WITH its
76
+ leading ``v`` (batch dirs are ``~/.g3dt/synth_metadata/v1.3.0/...``) — so
77
+ unlike ``_normalise_version`` nothing is stripped. A version that matches
78
+ no records is reported by the worker (skip + hint), not silently absorbed.
79
+ """
80
+ token = raw.strip()
81
+ return "all" if token.lower() == "all" else token
82
+
83
+
84
+ def _parse_study_specs(studies: str, fallback, env: str, synthetic: bool = False):
65
85
  """Turn ``--studies`` into ``[(resolved_study_key, version), ...]``.
66
86
 
67
87
  Each comma-separated entry is ``name`` or ``name:version``. A bare name
68
88
  takes *fallback* (the ``--version`` default); *fallback* is ``None`` when
69
- ``--version`` was not given, which makes a bare name a usage error.
89
+ ``--version`` was not given, which makes a bare name a usage error —
90
+ except with *synthetic*, where a bare name defaults to ``all`` (the whole
91
+ point of the flag is "wipe the synthetic project").
92
+
93
+ With *synthetic* the raw name IS the Gen3 project code: no study-registry
94
+ lookup, and version tokens pass through :func:`_synthetic_version`.
70
95
 
71
96
  Every entry is validated before anything is dispatched, so a typo in the
72
97
  last study cannot leave the earlier ones already deleted.
@@ -102,9 +127,15 @@ def _parse_study_specs(studies: str, fallback, env: str):
102
127
  raise typer.Exit(2)
103
128
 
104
129
  if sep:
105
- version = _normalise_version(raw_version, f"for study '{name}'")
130
+ version = (
131
+ _synthetic_version(raw_version)
132
+ if synthetic
133
+ else _normalise_version(raw_version, f"for study '{name}'")
134
+ )
106
135
  elif fallback is not None:
107
136
  version = fallback
137
+ elif synthetic:
138
+ version = "all"
108
139
  else:
109
140
  typer.secho(
110
141
  f"No version for study '{name}': add ':<version>' to it "
@@ -115,7 +146,7 @@ def _parse_study_specs(studies: str, fallback, env: str):
115
146
  )
116
147
  raise typer.Exit(2)
117
148
 
118
- specs.append((study_of(name, env).key, version))
149
+ specs.append((name if synthetic else study_of(name, env).key, version))
119
150
 
120
151
  if not specs:
121
152
  typer.secho("--studies is empty.", fg=typer.colors.RED, err=True)
@@ -141,6 +172,20 @@ def metadata(
141
172
  "':version', e.g. 0.9.8, or 'all' for every version.",
142
173
  ),
143
174
  node: Optional[str] = typer.Option(None, "--node", help="Delete only this node type."),
175
+ synthetic: bool = typer.Option(
176
+ False,
177
+ "--synthetic",
178
+ help="Registry-free synthetic-data mode: each --studies name is the "
179
+ "Gen3 project id itself (no SSM study registry). Bare names "
180
+ "default to version 'all'; a specific version matches records' "
181
+ "data_version property verbatim.",
182
+ ),
183
+ program_id: Optional[str] = typer.Option(
184
+ None,
185
+ "--program-id",
186
+ help="Gen3 program for --synthetic (default: program1). "
187
+ "Invalid without --synthetic.",
188
+ ),
144
189
  yes: bool = typer.Option(
145
190
  False, "--yes", "-y", help="Skip the non-prod prompt (specific-version only)."
146
191
  ),
@@ -156,12 +201,24 @@ def metadata(
156
201
 
157
202
  g3dt delete metadata --studies "ausdiab:0.7.5,cdah:0.8.1" --env staging
158
203
  g3dt delete metadata --studies "ausdiab:all,cdah" --version 0.9.8 --env staging
204
+ g3dt delete metadata --studies "synthetic_dataset_1,synthetic_dataset_2" --env test --synthetic
159
205
  """
160
206
  env = resolve.active_env(env)
161
- fallback = (
162
- _normalise_version(version, "for --version") if version is not None else None
163
- )
164
- specs = _parse_study_specs(studies, fallback, env)
207
+ if program_id is not None and not synthetic:
208
+ typer.secho(
209
+ "--program-id is only valid with --synthetic (registered studies "
210
+ "carry their program in the study registry).",
211
+ fg=typer.colors.RED,
212
+ err=True,
213
+ )
214
+ raise typer.Exit(2)
215
+ if version is None:
216
+ fallback = None
217
+ elif synthetic:
218
+ fallback = _synthetic_version(version)
219
+ else:
220
+ fallback = _normalise_version(version, "for --version")
221
+ specs = _parse_study_specs(studies, fallback, env, synthetic=synthetic)
165
222
  versions = [v for _, v in specs]
166
223
 
167
224
  # The typed production confirmation stays the study keys alone: short
@@ -171,13 +228,17 @@ def metadata(
171
228
  uniform = len(set(versions)) == 1
172
229
  any_all = "all" in versions
173
230
 
231
+ prefix = "synthetic " if synthetic else ""
174
232
  if uniform and versions[0] == "all":
175
- action = "deletion of ALL VERSIONS"
233
+ action = f"{prefix}deletion of ALL VERSIONS"
176
234
  elif uniform:
177
- action = f"deletion of v{versions[0]}"
235
+ # Synthetic versions are verbatim data_version values (often already
236
+ # v-prefixed); Athena versions are canonical x.y.z, displayed with v.
237
+ shown = versions[0] if synthetic else f"v{versions[0]}"
238
+ action = f"{prefix}deletion of {shown}"
178
239
  else:
179
240
  plan = ", ".join(f"{key}:{v}" for key, v in specs)
180
- action = f"deletion of per-study versions [{plan}]"
241
+ action = f"{prefix}deletion of per-study versions [{plan}]"
181
242
 
182
243
  # Deleting every version is the most destructive path: always prompt (pass
183
244
  # assume_yes=False so --yes can't bypass it; prod still types the target).
@@ -200,6 +261,11 @@ def metadata(
200
261
  ]
201
262
  if node:
202
263
  a += ["--node", node]
264
+ if synthetic:
265
+ # Program is always passed explicitly: the shell default makes it
266
+ # optional on the wire, but an explicit value keeps the contract
267
+ # visible in logs and SSM command history.
268
+ a += ["--synthetic", "--program-id", program_id or "program1"]
203
269
  return a
204
270
 
205
271
  def remote_cli(env_name):
@@ -212,6 +278,12 @@ def metadata(
212
278
  a.append("--yes")
213
279
  if node:
214
280
  a += ["--node", node]
281
+ if synthetic:
282
+ # Without this the remote re-entry would re-parse --studies
283
+ # against the study registry on the box and exit 2.
284
+ a.append("--synthetic")
285
+ if program_id is not None:
286
+ a += ["--program-id", program_id]
215
287
  return a
216
288
 
217
289
  dispatch.run_or_dispatch(
@@ -20,6 +20,7 @@ from g3dt.cli import (
20
20
  metadata,
21
21
  pipeline_cmds,
22
22
  release_cmds,
23
+ study_cmds,
23
24
  synth,
24
25
  )
25
26
 
@@ -49,6 +50,7 @@ def _root(
49
50
 
50
51
  app.add_typer(dict_cmds.app, name="dict")
51
52
  app.add_typer(synth.app, name="synth")
53
+ app.add_typer(study_cmds.app, name="study")
52
54
  app.add_typer(metadata.app, name="metadata")
53
55
  app.add_typer(delete_cmds.app, name="delete")
54
56
  app.add_typer(k8s.app, name="k8s")
@@ -64,12 +66,15 @@ _DOCS = """\
64
66
  Gen3 DataOps toolkit (g3dt) — operations overview
65
67
  =================================================
66
68
 
67
- Configuration: two kinds, nothing else
69
+ Configuration: three kinds, nothing else
68
70
  - INPUTS live in your deployment wrapper repo as config/<project>.<env>.json,
69
71
  read only by `cdk deploy` (via the aws-gen3-pipeline template).
70
- - Everything else is resolved live from SSM (/{project}/{env}/...), which
72
+ - OUTPUTS are resolved live from SSM (/{project}/{env}/...), which
71
73
  `cdk deploy` publishes. The only local file is the g3dt.yaml marker,
72
74
  searched at ./g3dt.yaml, ~/.g3dt/g3dt.yaml, /etc/g3dt/g3dt.yaml.
75
+ - OPERATIONAL state is written by the toolkit itself: the study registry
76
+ (SSM /{project}/{env}/studies/*, managed with `g3dt study`) and the
77
+ Iceberg ledgers (releases, upload receipts, indexd registry).
73
78
 
74
79
  Contexts: what am I pointed at?
75
80
  A context is a named (project, env, profile, region) tuple. Every command
@@ -97,14 +102,18 @@ Discover everything
97
102
  g3dt config contexts your contexts (current marked *; add
98
103
  --verify to check what is deployed)
99
104
  g3dt config envs environments with a deployed SSM tree
100
- g3dt config studies studies from your g3dt.yaml marker
105
+ g3dt study list the env's study registry (config studies
106
+ is an alias)
101
107
  g3dt config show resolved settings for the current context
102
108
 
103
109
  Typical release runbook (staging shown; repeat for prod with care)
104
110
  1. g3dt dict deploy --env staging
105
- 2. g3dt metadata upload --study <study> --env staging --on ec2
106
- 3. g3dt jobs logs <run-id> --follow
107
- 4. g3dt k8s restart-etl --env staging
111
+ 2. g3dt study repoint --latest --env staging point the registry at the
112
+ newest release (validates
113
+ every target first)
114
+ 3. g3dt metadata upload --study <study> --env staging --on ec2
115
+ 4. g3dt jobs logs <run-id> --follow
116
+ 5. g3dt k8s restart-etl --env staging
108
117
 
109
118
  Promoting one dictionary across environments
110
119
  The source repo/path are env inputs; usually only the tag changes, and it