gen3-dataops-toolkit 3.5.0__tar.gz → 3.6.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/PKG-INFO +23 -1
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/README.md +22 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/pyproject.toml +1 -1
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/synth.py +82 -20
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/config.py +16 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/synthetic_data/full_deploy_dd_and_synth.sh +21 -6
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/__init__.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/__init__.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/_internal/__init__.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/_internal/dispatch.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/_internal/registry.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/_internal/resolve.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/_internal/runner.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/_internal/safety.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/config_cmds.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/delete_cmds.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/dict_cmds.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/ec2_cmds.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/indexd_cmds.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/jobs.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/k8s.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/main.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/metadata.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/pipeline_cmds.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/release_cmds.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/indexd/__init__.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/indexd/file_access.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/indexd/indexd_registrar.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/ingest/ingest.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/resolver.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/delete/delete_all_metadata_for_project.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/delete/delete_metadata.sh +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/delete/delete_metadata_by_guid.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/dictionary/deploy_dd.sh +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/dictionary/pull_dict.sh +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/dictionary/upload_dictionary.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/indexd/register_indexd.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/indexd/verify_file_access.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/k8s_ops/argocd_restart_etl.sh +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/k8s_ops/argocd_restart_ms.sh +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/k8s_ops/argocd_restart_schema.sh +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/k8s_ops/login_to_pod.sh +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/k8s_ops/restart_etl_and_ms.sh +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/synthetic_data/delete_synth_metadata_sheepdog.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/synthetic_data/generate_synth_metadata.sh +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/synthetic_data/upload_synth_metadata_sheepdog.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/upload/metadata/upload_all_studies.sh +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/upload/metadata/upload_metadata.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/upload/__init__.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/upload/metadata_deleter.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/upload/metadata_submitter.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/upload/upload_synthdata_s3.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/utils/athena_utils.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/utils/dbt_utils.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/utils/release_writer.py +0 -0
- {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/validate/validate.py +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: gen3-dataops-toolkit
|
|
3
|
-
Version: 3.
|
|
3
|
+
Version: 3.6.0
|
|
4
4
|
Summary: Gen3 DataOps toolkit (g3dt): operate SSM-published Gen3 data pipeline environments
|
|
5
5
|
License: Apache-2.0
|
|
6
6
|
Author: JoshuaHarris391
|
|
@@ -208,6 +208,28 @@ g3dt synth generate AusDiab_Simulated --llm -n 5 -e test
|
|
|
208
208
|
`g3dt config show --env <env>` prints the resolved provider, model, and key
|
|
209
209
|
path.
|
|
210
210
|
|
|
211
|
+
### The synthetic data lifecycle
|
|
212
|
+
|
|
213
|
+
`synth generate` builds a batch locally (keyless `random` by default, `--llm`
|
|
214
|
+
for realistic values); `synth upload` pushes a batch to the commons; `synth
|
|
215
|
+
delete` removes one; and `synth deploy` runs the whole cycle in one command —
|
|
216
|
+
dictionary pull + upload, schema restarts, delete of the previous batch,
|
|
217
|
+
LLM-generate, upload, and the ETL run.
|
|
218
|
+
|
|
219
|
+
**Studies and record counts are batch inputs, not environment facts** — they
|
|
220
|
+
are passed per command and never come from SSM:
|
|
221
|
+
|
|
222
|
+
```bash
|
|
223
|
+
g3dt synth generate synthetic_dataset_1 --llm -n 100 -e test
|
|
224
|
+
g3dt synth deploy -e test --studies synthetic_dataset_1 -n 100 --prev-version v1.2.0
|
|
225
|
+
```
|
|
226
|
+
|
|
227
|
+
`deploy` without `--studies` falls back to the original ACDC demo set
|
|
228
|
+
(`AusDiab_Simulated,Baker-Biobank_Simulated,BioHeart-CT_Simulated,CAUGHT-CAD_Simulated`
|
|
229
|
+
at 30,60,20,55 records, previous batch `v1.0.0`) — kept for continuity; any
|
|
230
|
+
other project should always pass its own `--studies`. The deploy's delete step
|
|
231
|
+
targets exactly the studies being regenerated, at `--prev-version`.
|
|
232
|
+
|
|
211
233
|
### Kubernetes restart targets
|
|
212
234
|
|
|
213
235
|
`k8s restart-schema`, `k8s restart-ms`, `dict deploy`, and `synth deploy`
|
|
@@ -174,6 +174,28 @@ g3dt synth generate AusDiab_Simulated --llm -n 5 -e test
|
|
|
174
174
|
`g3dt config show --env <env>` prints the resolved provider, model, and key
|
|
175
175
|
path.
|
|
176
176
|
|
|
177
|
+
### The synthetic data lifecycle
|
|
178
|
+
|
|
179
|
+
`synth generate` builds a batch locally (keyless `random` by default, `--llm`
|
|
180
|
+
for realistic values); `synth upload` pushes a batch to the commons; `synth
|
|
181
|
+
delete` removes one; and `synth deploy` runs the whole cycle in one command —
|
|
182
|
+
dictionary pull + upload, schema restarts, delete of the previous batch,
|
|
183
|
+
LLM-generate, upload, and the ETL run.
|
|
184
|
+
|
|
185
|
+
**Studies and record counts are batch inputs, not environment facts** — they
|
|
186
|
+
are passed per command and never come from SSM:
|
|
187
|
+
|
|
188
|
+
```bash
|
|
189
|
+
g3dt synth generate synthetic_dataset_1 --llm -n 100 -e test
|
|
190
|
+
g3dt synth deploy -e test --studies synthetic_dataset_1 -n 100 --prev-version v1.2.0
|
|
191
|
+
```
|
|
192
|
+
|
|
193
|
+
`deploy` without `--studies` falls back to the original ACDC demo set
|
|
194
|
+
(`AusDiab_Simulated,Baker-Biobank_Simulated,BioHeart-CT_Simulated,CAUGHT-CAD_Simulated`
|
|
195
|
+
at 30,60,20,55 records, previous batch `v1.0.0`) — kept for continuity; any
|
|
196
|
+
other project should always pass its own `--studies`. The deploy's delete step
|
|
197
|
+
targets exactly the studies being regenerated, at `--prev-version`.
|
|
198
|
+
|
|
177
199
|
### Kubernetes restart targets
|
|
178
200
|
|
|
179
201
|
`k8s restart-schema`, `k8s restart-ms`, `dict deploy`, and `synth deploy`
|
|
@@ -9,6 +9,12 @@ Every command accepts ``--env``; targeting a **production** environment (any env
|
|
|
9
9
|
whose name contains ``prod``) shows a warning and requires typing the env name to
|
|
10
10
|
confirm — it cannot be bypassed.
|
|
11
11
|
|
|
12
|
+
Studies and record counts are **batch inputs**, passed per command
|
|
13
|
+
(``generate STUDIES -n N``, ``deploy --studies ... -n ...``) — they are not
|
|
14
|
+
environment facts, so they never come from SSM. The ``deploy`` defaults are
|
|
15
|
+
the original ACDC demo set, kept for continuity; other projects pass their
|
|
16
|
+
own.
|
|
17
|
+
|
|
12
18
|
Generation defaults to keyless ``random`` data (no API calls). Pass ``--llm``
|
|
13
19
|
for LLM-realistic values. The LLM provider and model come from the
|
|
14
20
|
environment's SSM tree (the CDK config's optional ``llm`` block, published as
|
|
@@ -44,6 +50,32 @@ app = typer.Typer(
|
|
|
44
50
|
|
|
45
51
|
SYNTH_DIR = Path("~/.g3dt/synth_metadata").expanduser()
|
|
46
52
|
|
|
53
|
+
#: Defaults for the full-deploy flow, kept for continuity with the original
|
|
54
|
+
#: ACDC demo environment. Any other project should pass --studies (and
|
|
55
|
+
#: --num-records) — there is no SSM fact for these because simulated study
|
|
56
|
+
#: sets are batch inputs, not environment facts.
|
|
57
|
+
DEPLOY_DEFAULT_STUDIES = (
|
|
58
|
+
"AusDiab_Simulated,Baker-Biobank_Simulated,"
|
|
59
|
+
"BioHeart-CT_Simulated,CAUGHT-CAD_Simulated"
|
|
60
|
+
)
|
|
61
|
+
DEPLOY_DEFAULT_NUM_RECORDS = "30,60,20,55"
|
|
62
|
+
DEPLOY_DEFAULT_PREV_VERSION = "v1.0.0"
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def _check_per_study_counts(studies: str, num_records: Optional[str]) -> None:
|
|
66
|
+
"""Exit 1 when a per-study count list does not line up with the studies."""
|
|
67
|
+
if num_records and "," in num_records:
|
|
68
|
+
n_counts = len(num_records.split(","))
|
|
69
|
+
n_studies = len(studies.split(","))
|
|
70
|
+
if n_counts != n_studies:
|
|
71
|
+
typer.secho(
|
|
72
|
+
f"--num-records has {n_counts} values but {n_studies} studies "
|
|
73
|
+
f"were given (pass one count, or one per study).",
|
|
74
|
+
fg=typer.colors.RED,
|
|
75
|
+
err=True,
|
|
76
|
+
)
|
|
77
|
+
raise typer.Exit(1)
|
|
78
|
+
|
|
47
79
|
|
|
48
80
|
def _llm_env_overrides(
|
|
49
81
|
e,
|
|
@@ -177,22 +209,62 @@ def deploy(
|
|
|
177
209
|
None, "--etl-cronjob",
|
|
178
210
|
help="ETL cronjob name; default: the env's SSM app/etl_cronjob.",
|
|
179
211
|
),
|
|
212
|
+
studies: Optional[str] = typer.Option(
|
|
213
|
+
None, "--studies",
|
|
214
|
+
help="Simulated study id(s), comma-separated. These are (re)generated, "
|
|
215
|
+
"uploaded, and their previous batch deleted. Default: the original "
|
|
216
|
+
f"demo set ({DEPLOY_DEFAULT_STUDIES}).",
|
|
217
|
+
),
|
|
218
|
+
num_records: Optional[str] = typer.Option(
|
|
219
|
+
None, "--num-records", "-n",
|
|
220
|
+
help="Records per study: one number for all, or a comma list (one per "
|
|
221
|
+
"study). Default: 30 per study (or the classic 30,60,20,55 when "
|
|
222
|
+
"--studies is not given).",
|
|
223
|
+
),
|
|
224
|
+
prev_version: Optional[str] = typer.Option(
|
|
225
|
+
None, "--prev-version",
|
|
226
|
+
help="Dictionary version whose previously-uploaded synthetic batch is "
|
|
227
|
+
f"deleted before uploading the new one. Default: {DEPLOY_DEFAULT_PREV_VERSION}.",
|
|
228
|
+
),
|
|
180
229
|
) -> None:
|
|
181
|
-
"""Full end-to-end synthetic deploy
|
|
230
|
+
"""Full end-to-end synthetic deploy: the whole cycle in one command.
|
|
231
|
+
|
|
232
|
+
Wraps services/synthetic_data/full_deploy_dd_and_synth.sh, which runs:
|
|
233
|
+
|
|
234
|
+
\b
|
|
235
|
+
1. pull the dictionary at the env's version and upload it to S3
|
|
236
|
+
2. restart the schema microservices (env's SSM restart_services order)
|
|
237
|
+
3. delete the PREVIOUS synthetic batch for the given studies
|
|
238
|
+
(--prev-version, so stale records don't linger in the commons)
|
|
239
|
+
4. LLM-generate a new batch for the studies (provider/model from SSM)
|
|
240
|
+
5. upload the new batch to Gen3
|
|
241
|
+
6. run the ETL cronjob (env's SSM etl_cronjob)
|
|
182
242
|
|
|
183
|
-
|
|
184
|
-
generation). Provider/model and the restart targets come from the env's
|
|
243
|
+
Provider/model, restart targets, and the ETL cronjob come from the env's
|
|
185
244
|
SSM tree unless overridden; the API key path comes from
|
|
186
|
-
--llm-api-key-file or the marker.
|
|
245
|
+
--llm-api-key-file or the marker. Studies and record counts are batch
|
|
246
|
+
inputs with ACDC-era defaults — any other project should pass --studies.
|
|
247
|
+
|
|
248
|
+
Examples:
|
|
249
|
+
g3dt synth deploy -e test --studies synthetic_dataset_1 -n 100
|
|
250
|
+
g3dt synth deploy -e test --studies "s1,s2" -n "100,50" --prev-version v1.2.0
|
|
187
251
|
"""
|
|
188
252
|
e = env_of(env)
|
|
189
253
|
safety.confirm_prod_strict("synthetic full deploy", env)
|
|
254
|
+
effective_studies = studies or DEPLOY_DEFAULT_STUDIES
|
|
255
|
+
_check_per_study_counts(effective_studies, num_records)
|
|
190
256
|
env_vars = script_env(e)
|
|
191
257
|
env_vars.update(_llm_env_overrides(e, llm_provider, llm_model, llm_api_key_file))
|
|
192
258
|
if restart_services:
|
|
193
259
|
env_vars["G3DT_RESTART_SERVICES"] = restart_services
|
|
194
260
|
if etl_cronjob:
|
|
195
261
|
env_vars["G3DT_ETL_CRONJOB"] = etl_cronjob
|
|
262
|
+
if studies:
|
|
263
|
+
env_vars["G3DT_SYNTH_STUDIES"] = studies
|
|
264
|
+
if num_records:
|
|
265
|
+
env_vars["G3DT_SYNTH_NUM_RECORDS"] = num_records
|
|
266
|
+
if prev_version:
|
|
267
|
+
env_vars["G3DT_SYNTH_PREV_VERSION"] = prev_version
|
|
196
268
|
runner.run(
|
|
197
269
|
runner.bash_script(
|
|
198
270
|
"services/synthetic_data/full_deploy_dd_and_synth.sh", env
|
|
@@ -254,25 +326,15 @@ def generate(
|
|
|
254
326
|
generate LLM-realistic values instead.
|
|
255
327
|
|
|
256
328
|
Examples:
|
|
257
|
-
g3dt synth generate
|
|
258
|
-
g3dt synth generate
|
|
259
|
-
g3dt synth generate "
|
|
329
|
+
g3dt synth generate synthetic_dataset_1 -n 5 --seed 1
|
|
330
|
+
g3dt synth generate synthetic_dataset_1 --llm -n 100
|
|
331
|
+
g3dt synth generate "dataset_a,dataset_b" -n "30,60"
|
|
332
|
+
g3dt synth generate dataset_a --llm --llm-model gpt-4o-mini \
|
|
333
|
+
--llm-api-key-file ~/keys/openai_api_key.txt
|
|
260
334
|
"""
|
|
261
335
|
e = env_of(env)
|
|
262
336
|
safety.confirm_prod_strict("synthetic generation", env)
|
|
263
|
-
|
|
264
|
-
# A comma list of per-study counts must line up with the studies given.
|
|
265
|
-
if num_records and "," in num_records:
|
|
266
|
-
n_counts = len(num_records.split(","))
|
|
267
|
-
n_studies = len(studies.split(","))
|
|
268
|
-
if n_counts != n_studies:
|
|
269
|
-
typer.secho(
|
|
270
|
-
f"--num-records has {n_counts} values but {n_studies} studies "
|
|
271
|
-
f"were given (pass one count, or one per study).",
|
|
272
|
-
fg=typer.colors.RED,
|
|
273
|
-
err=True,
|
|
274
|
-
)
|
|
275
|
-
raise typer.Exit(1)
|
|
337
|
+
_check_per_study_counts(studies, num_records)
|
|
276
338
|
|
|
277
339
|
ver = version or e.dictionary_version
|
|
278
340
|
schema_path = schema or str(SCHEMA_DIR / dictionary_filename(e, ver))
|
|
@@ -428,8 +428,24 @@ def script_env(e: EnvConfig, version: Optional[str] = None) -> Dict[str, str]:
|
|
|
428
428
|
scripts compose a URL keeps Python the single source of truth: the scripts
|
|
429
429
|
read ``G3DT_DICT_URL``/``G3DT_DICT_FILENAME`` and never build either
|
|
430
430
|
themselves, so there is only one implementation to keep correct.
|
|
431
|
+
|
|
432
|
+
The interpreter's own bin directory is prepended to PATH: console scripts
|
|
433
|
+
installed next to g3dt — gen3-metadata-simulator via
|
|
434
|
+
``g3dt synth install-simulator`` — land there, but a pipx install exposes
|
|
435
|
+
only g3dt's own entry points on the caller's PATH, so without this the
|
|
436
|
+
wrapped scripts' ``command -v gen3-metadata-simulator`` check fails even
|
|
437
|
+
though the tool is installed.
|
|
431
438
|
"""
|
|
439
|
+
import sys
|
|
440
|
+
|
|
432
441
|
env = dict(os.environ)
|
|
442
|
+
# No .resolve(): a venv's python is a symlink to the base interpreter, and
|
|
443
|
+
# resolving it would point at the base install's bin instead of the venv
|
|
444
|
+
# bin where console scripts are actually created.
|
|
445
|
+
venv_bin = str(Path(sys.executable).parent)
|
|
446
|
+
path_entries = env.get("PATH", "").split(os.pathsep)
|
|
447
|
+
if venv_bin not in path_entries:
|
|
448
|
+
env["PATH"] = venv_bin + os.pathsep + env.get("PATH", "")
|
|
433
449
|
values = {
|
|
434
450
|
"G3DT_ENV": e.name,
|
|
435
451
|
"G3DT_REGION": e.region,
|
|
@@ -43,10 +43,24 @@ ARGO_SCRIPT_DIR="${SERVICE_DIR}/k8s_ops"
|
|
|
43
43
|
SCHEMA_DIR="${G3DT_SCHEMA_DIR:-$HOME/.g3dt/schemas}"
|
|
44
44
|
SYNTH_BASE="${G3DT_SYNTH_DIR:-$HOME/.g3dt/synth_metadata}"
|
|
45
45
|
|
|
46
|
+
# Batch inputs (set by `g3dt synth deploy --studies/--num-records/--prev-version`;
|
|
47
|
+
# the fallbacks are the original ACDC demo values, kept for continuity).
|
|
48
|
+
STUDIES="${G3DT_SYNTH_STUDIES:-AusDiab_Simulated,Baker-Biobank_Simulated,BioHeart-CT_Simulated,CAUGHT-CAD_Simulated}"
|
|
49
|
+
if [ -n "${G3DT_SYNTH_NUM_RECORDS:-}" ]; then
|
|
50
|
+
NUM_RECORDS="${G3DT_SYNTH_NUM_RECORDS}"
|
|
51
|
+
elif [ -n "${G3DT_SYNTH_STUDIES:-}" ]; then
|
|
52
|
+
NUM_RECORDS="30" # custom studies, no counts given: 30 per study
|
|
53
|
+
else
|
|
54
|
+
NUM_RECORDS="30,60,20,55" # the classic per-study counts for the demo set
|
|
55
|
+
fi
|
|
56
|
+
PREV_VERSION="${G3DT_SYNTH_PREV_VERSION:-v1.0.0}"
|
|
57
|
+
|
|
46
58
|
# Derived variables
|
|
47
|
-
PREV_VERSION="v1.0.0"
|
|
48
59
|
SYNTH_META_DIR="${SYNTH_BASE}/${VERSION}/"
|
|
49
|
-
|
|
60
|
+
# The delete step walks the previous batch's import order; every study in a
|
|
61
|
+
# batch shares one, so the first study's copy serves.
|
|
62
|
+
FIRST_STUDY="${STUDIES%%,*}"
|
|
63
|
+
DATA_IMPORT_ORDER_FILE="${SYNTH_BASE}/${PREV_VERSION}/${FIRST_STUDY}/DataImportOrder.txt"
|
|
50
64
|
|
|
51
65
|
# Never export an empty AWS_PROFILE (empty means ambient credentials).
|
|
52
66
|
if [ -n "${G3DT_AWS_PROFILE:-}" ]; then
|
|
@@ -87,9 +101,9 @@ bash "${ARGO_SCRIPT_DIR}/argocd_restart_schema.sh" \
|
|
|
87
101
|
-a "${APP_NAME}" \
|
|
88
102
|
-n "${NAMESPACE}"
|
|
89
103
|
|
|
90
|
-
echo "==== [4] Deleting old synthetic data for version ${PREV_VERSION} ===="
|
|
104
|
+
echo "==== [4] Deleting old synthetic data (${STUDIES}) for version ${PREV_VERSION} ===="
|
|
91
105
|
DELETE_SYNTH_ARGS=(
|
|
92
|
-
-p "
|
|
106
|
+
-p "${STUDIES}"
|
|
93
107
|
-s "${AWS_SECRET_NAME}"
|
|
94
108
|
-i "${DATA_IMPORT_ORDER_FILE}"
|
|
95
109
|
)
|
|
@@ -98,12 +112,13 @@ if [ -n "${G3DT_AWS_PROFILE:-}" ]; then
|
|
|
98
112
|
fi
|
|
99
113
|
python3 "${SERVICE_DIR}/synthetic_data/delete_synth_metadata_sheepdog.py" "${DELETE_SYNTH_ARGS[@]}"
|
|
100
114
|
|
|
101
|
-
echo "==== [5] Generating new synthetic data for version ${VERSION} (LLM-realistic) ===="
|
|
115
|
+
echo "==== [5] Generating new synthetic data (${STUDIES}) for version ${VERSION} (LLM-realistic) ===="
|
|
102
116
|
bash "${SERVICE_DIR}/synthetic_data/generate_synth_metadata.sh" \
|
|
103
117
|
--schema "${SCHEMA_DIR}/${DICT_FILENAME}" \
|
|
104
118
|
--version "${VERSION}" \
|
|
105
119
|
--provider llm \
|
|
106
|
-
--
|
|
120
|
+
--studies "${STUDIES}" \
|
|
121
|
+
--num-records "${NUM_RECORDS}" \
|
|
107
122
|
--output-root "${SYNTH_BASE}"
|
|
108
123
|
|
|
109
124
|
echo "==== [6] Uploading new synthetic data for version ${VERSION} ===="
|
|
File without changes
|
|
File without changes
|
{gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/_internal/__init__.py
RENAMED
|
File without changes
|
{gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/_internal/dispatch.py
RENAMED
|
File without changes
|
{gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/_internal/registry.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/indexd/indexd_registrar.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/dictionary/deploy_dd.sh
RENAMED
|
File without changes
|
{gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/dictionary/pull_dict.sh
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/k8s_ops/login_to_pod.sh
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
{gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/upload/metadata_deleter.py
RENAMED
|
File without changes
|
{gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/upload/metadata_submitter.py
RENAMED
|
File without changes
|
{gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/upload/upload_synthdata_s3.py
RENAMED
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|