gen3-dataops-toolkit 3.5.0__tar.gz → 3.6.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (56) hide show
  1. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/PKG-INFO +23 -1
  2. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/README.md +22 -0
  3. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/pyproject.toml +1 -1
  4. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/synth.py +82 -20
  5. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/config.py +16 -0
  6. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/synthetic_data/full_deploy_dd_and_synth.sh +21 -6
  7. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/__init__.py +0 -0
  8. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/__init__.py +0 -0
  9. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/_internal/__init__.py +0 -0
  10. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/_internal/dispatch.py +0 -0
  11. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/_internal/registry.py +0 -0
  12. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/_internal/resolve.py +0 -0
  13. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/_internal/runner.py +0 -0
  14. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/_internal/safety.py +0 -0
  15. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/config_cmds.py +0 -0
  16. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/delete_cmds.py +0 -0
  17. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/dict_cmds.py +0 -0
  18. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/ec2_cmds.py +0 -0
  19. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/indexd_cmds.py +0 -0
  20. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/jobs.py +0 -0
  21. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/k8s.py +0 -0
  22. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/main.py +0 -0
  23. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/metadata.py +0 -0
  24. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/pipeline_cmds.py +0 -0
  25. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/cli/release_cmds.py +0 -0
  26. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/indexd/__init__.py +0 -0
  27. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/indexd/file_access.py +0 -0
  28. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/indexd/indexd_registrar.py +0 -0
  29. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/ingest/ingest.py +0 -0
  30. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/resolver.py +0 -0
  31. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/delete/delete_all_metadata_for_project.py +0 -0
  32. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/delete/delete_metadata.sh +0 -0
  33. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/delete/delete_metadata_by_guid.py +0 -0
  34. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/dictionary/deploy_dd.sh +0 -0
  35. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/dictionary/pull_dict.sh +0 -0
  36. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/dictionary/upload_dictionary.py +0 -0
  37. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/indexd/register_indexd.py +0 -0
  38. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/indexd/verify_file_access.py +0 -0
  39. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/k8s_ops/argocd_restart_etl.sh +0 -0
  40. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/k8s_ops/argocd_restart_ms.sh +0 -0
  41. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/k8s_ops/argocd_restart_schema.sh +0 -0
  42. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/k8s_ops/login_to_pod.sh +0 -0
  43. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/k8s_ops/restart_etl_and_ms.sh +0 -0
  44. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/synthetic_data/delete_synth_metadata_sheepdog.py +0 -0
  45. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/synthetic_data/generate_synth_metadata.sh +0 -0
  46. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/synthetic_data/upload_synth_metadata_sheepdog.py +0 -0
  47. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/upload/metadata/upload_all_studies.sh +0 -0
  48. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/services/upload/metadata/upload_metadata.py +0 -0
  49. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/upload/__init__.py +0 -0
  50. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/upload/metadata_deleter.py +0 -0
  51. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/upload/metadata_submitter.py +0 -0
  52. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/upload/upload_synthdata_s3.py +0 -0
  53. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/utils/athena_utils.py +0 -0
  54. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/utils/dbt_utils.py +0 -0
  55. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/utils/release_writer.py +0 -0
  56. {gen3_dataops_toolkit-3.5.0 → gen3_dataops_toolkit-3.6.0}/src/g3dt/validate/validate.py +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: gen3-dataops-toolkit
3
- Version: 3.5.0
3
+ Version: 3.6.0
4
4
  Summary: Gen3 DataOps toolkit (g3dt): operate SSM-published Gen3 data pipeline environments
5
5
  License: Apache-2.0
6
6
  Author: JoshuaHarris391
@@ -208,6 +208,28 @@ g3dt synth generate AusDiab_Simulated --llm -n 5 -e test
208
208
  `g3dt config show --env <env>` prints the resolved provider, model, and key
209
209
  path.
210
210
 
211
+ ### The synthetic data lifecycle
212
+
213
+ `synth generate` builds a batch locally (keyless `random` by default, `--llm`
214
+ for realistic values); `synth upload` pushes a batch to the commons; `synth
215
+ delete` removes one; and `synth deploy` runs the whole cycle in one command —
216
+ dictionary pull + upload, schema restarts, delete of the previous batch,
217
+ LLM-generate, upload, and the ETL run.
218
+
219
+ **Studies and record counts are batch inputs, not environment facts** — they
220
+ are passed per command and never come from SSM:
221
+
222
+ ```bash
223
+ g3dt synth generate synthetic_dataset_1 --llm -n 100 -e test
224
+ g3dt synth deploy -e test --studies synthetic_dataset_1 -n 100 --prev-version v1.2.0
225
+ ```
226
+
227
+ `deploy` without `--studies` falls back to the original ACDC demo set
228
+ (`AusDiab_Simulated,Baker-Biobank_Simulated,BioHeart-CT_Simulated,CAUGHT-CAD_Simulated`
229
+ at 30,60,20,55 records, previous batch `v1.0.0`) — kept for continuity; any
230
+ other project should always pass its own `--studies`. The deploy's delete step
231
+ targets exactly the studies being regenerated, at `--prev-version`.
232
+
211
233
  ### Kubernetes restart targets
212
234
 
213
235
  `k8s restart-schema`, `k8s restart-ms`, `dict deploy`, and `synth deploy`
@@ -174,6 +174,28 @@ g3dt synth generate AusDiab_Simulated --llm -n 5 -e test
174
174
  `g3dt config show --env <env>` prints the resolved provider, model, and key
175
175
  path.
176
176
 
177
+ ### The synthetic data lifecycle
178
+
179
+ `synth generate` builds a batch locally (keyless `random` by default, `--llm`
180
+ for realistic values); `synth upload` pushes a batch to the commons; `synth
181
+ delete` removes one; and `synth deploy` runs the whole cycle in one command —
182
+ dictionary pull + upload, schema restarts, delete of the previous batch,
183
+ LLM-generate, upload, and the ETL run.
184
+
185
+ **Studies and record counts are batch inputs, not environment facts** — they
186
+ are passed per command and never come from SSM:
187
+
188
+ ```bash
189
+ g3dt synth generate synthetic_dataset_1 --llm -n 100 -e test
190
+ g3dt synth deploy -e test --studies synthetic_dataset_1 -n 100 --prev-version v1.2.0
191
+ ```
192
+
193
+ `deploy` without `--studies` falls back to the original ACDC demo set
194
+ (`AusDiab_Simulated,Baker-Biobank_Simulated,BioHeart-CT_Simulated,CAUGHT-CAD_Simulated`
195
+ at 30,60,20,55 records, previous batch `v1.0.0`) — kept for continuity; any
196
+ other project should always pass its own `--studies`. The deploy's delete step
197
+ targets exactly the studies being regenerated, at `--prev-version`.
198
+
177
199
  ### Kubernetes restart targets
178
200
 
179
201
  `k8s restart-schema`, `k8s restart-ms`, `dict deploy`, and `synth deploy`
@@ -1,6 +1,6 @@
1
1
  [tool.poetry]
2
2
  name = "gen3-dataops-toolkit"
3
- version = "3.5.0"
3
+ version = "3.6.0"
4
4
  description = "Gen3 DataOps toolkit (g3dt): operate SSM-published Gen3 data pipeline environments"
5
5
  authors = ["JoshuaHarris391 <harjo391@gmail.com>"]
6
6
  readme = "README.md"
@@ -9,6 +9,12 @@ Every command accepts ``--env``; targeting a **production** environment (any env
9
9
  whose name contains ``prod``) shows a warning and requires typing the env name to
10
10
  confirm — it cannot be bypassed.
11
11
 
12
+ Studies and record counts are **batch inputs**, passed per command
13
+ (``generate STUDIES -n N``, ``deploy --studies ... -n ...``) — they are not
14
+ environment facts, so they never come from SSM. The ``deploy`` defaults are
15
+ the original ACDC demo set, kept for continuity; other projects pass their
16
+ own.
17
+
12
18
  Generation defaults to keyless ``random`` data (no API calls). Pass ``--llm``
13
19
  for LLM-realistic values. The LLM provider and model come from the
14
20
  environment's SSM tree (the CDK config's optional ``llm`` block, published as
@@ -44,6 +50,32 @@ app = typer.Typer(
44
50
 
45
51
  SYNTH_DIR = Path("~/.g3dt/synth_metadata").expanduser()
46
52
 
53
+ #: Defaults for the full-deploy flow, kept for continuity with the original
54
+ #: ACDC demo environment. Any other project should pass --studies (and
55
+ #: --num-records) — there is no SSM fact for these because simulated study
56
+ #: sets are batch inputs, not environment facts.
57
+ DEPLOY_DEFAULT_STUDIES = (
58
+ "AusDiab_Simulated,Baker-Biobank_Simulated,"
59
+ "BioHeart-CT_Simulated,CAUGHT-CAD_Simulated"
60
+ )
61
+ DEPLOY_DEFAULT_NUM_RECORDS = "30,60,20,55"
62
+ DEPLOY_DEFAULT_PREV_VERSION = "v1.0.0"
63
+
64
+
65
+ def _check_per_study_counts(studies: str, num_records: Optional[str]) -> None:
66
+ """Exit 1 when a per-study count list does not line up with the studies."""
67
+ if num_records and "," in num_records:
68
+ n_counts = len(num_records.split(","))
69
+ n_studies = len(studies.split(","))
70
+ if n_counts != n_studies:
71
+ typer.secho(
72
+ f"--num-records has {n_counts} values but {n_studies} studies "
73
+ f"were given (pass one count, or one per study).",
74
+ fg=typer.colors.RED,
75
+ err=True,
76
+ )
77
+ raise typer.Exit(1)
78
+
47
79
 
48
80
  def _llm_env_overrides(
49
81
  e,
@@ -177,22 +209,62 @@ def deploy(
177
209
  None, "--etl-cronjob",
178
210
  help="ETL cronjob name; default: the env's SSM app/etl_cronjob.",
179
211
  ),
212
+ studies: Optional[str] = typer.Option(
213
+ None, "--studies",
214
+ help="Simulated study id(s), comma-separated. These are (re)generated, "
215
+ "uploaded, and their previous batch deleted. Default: the original "
216
+ f"demo set ({DEPLOY_DEFAULT_STUDIES}).",
217
+ ),
218
+ num_records: Optional[str] = typer.Option(
219
+ None, "--num-records", "-n",
220
+ help="Records per study: one number for all, or a comma list (one per "
221
+ "study). Default: 30 per study (or the classic 30,60,20,55 when "
222
+ "--studies is not given).",
223
+ ),
224
+ prev_version: Optional[str] = typer.Option(
225
+ None, "--prev-version",
226
+ help="Dictionary version whose previously-uploaded synthetic batch is "
227
+ f"deleted before uploading the new one. Default: {DEPLOY_DEFAULT_PREV_VERSION}.",
228
+ ),
180
229
  ) -> None:
181
- """Full end-to-end synthetic deploy (dict + LLM-generate + upload + restarts).
230
+ """Full end-to-end synthetic deploy: the whole cycle in one command.
231
+
232
+ Wraps services/synthetic_data/full_deploy_dd_and_synth.sh, which runs:
233
+
234
+ \b
235
+ 1. pull the dictionary at the env's version and upload it to S3
236
+ 2. restart the schema microservices (env's SSM restart_services order)
237
+ 3. delete the PREVIOUS synthetic batch for the given studies
238
+ (--prev-version, so stale records don't linger in the commons)
239
+ 4. LLM-generate a new batch for the studies (provider/model from SSM)
240
+ 5. upload the new batch to Gen3
241
+ 6. run the ETL cronjob (env's SSM etl_cronjob)
182
242
 
183
- Wraps services/synthetic_data/full_deploy_dd_and_synth.sh (LLM-backed
184
- generation). Provider/model and the restart targets come from the env's
243
+ Provider/model, restart targets, and the ETL cronjob come from the env's
185
244
  SSM tree unless overridden; the API key path comes from
186
- --llm-api-key-file or the marker.
245
+ --llm-api-key-file or the marker. Studies and record counts are batch
246
+ inputs with ACDC-era defaults — any other project should pass --studies.
247
+
248
+ Examples:
249
+ g3dt synth deploy -e test --studies synthetic_dataset_1 -n 100
250
+ g3dt synth deploy -e test --studies "s1,s2" -n "100,50" --prev-version v1.2.0
187
251
  """
188
252
  e = env_of(env)
189
253
  safety.confirm_prod_strict("synthetic full deploy", env)
254
+ effective_studies = studies or DEPLOY_DEFAULT_STUDIES
255
+ _check_per_study_counts(effective_studies, num_records)
190
256
  env_vars = script_env(e)
191
257
  env_vars.update(_llm_env_overrides(e, llm_provider, llm_model, llm_api_key_file))
192
258
  if restart_services:
193
259
  env_vars["G3DT_RESTART_SERVICES"] = restart_services
194
260
  if etl_cronjob:
195
261
  env_vars["G3DT_ETL_CRONJOB"] = etl_cronjob
262
+ if studies:
263
+ env_vars["G3DT_SYNTH_STUDIES"] = studies
264
+ if num_records:
265
+ env_vars["G3DT_SYNTH_NUM_RECORDS"] = num_records
266
+ if prev_version:
267
+ env_vars["G3DT_SYNTH_PREV_VERSION"] = prev_version
196
268
  runner.run(
197
269
  runner.bash_script(
198
270
  "services/synthetic_data/full_deploy_dd_and_synth.sh", env
@@ -254,25 +326,15 @@ def generate(
254
326
  generate LLM-realistic values instead.
255
327
 
256
328
  Examples:
257
- g3dt synth generate AusDiab_Simulated -n 5 --seed 1
258
- g3dt synth generate AusDiab_Simulated --llm -n 5
259
- g3dt synth generate "AusDiab_Simulated,Baker-Biobank_Simulated" -n "30,60"
329
+ g3dt synth generate synthetic_dataset_1 -n 5 --seed 1
330
+ g3dt synth generate synthetic_dataset_1 --llm -n 100
331
+ g3dt synth generate "dataset_a,dataset_b" -n "30,60"
332
+ g3dt synth generate dataset_a --llm --llm-model gpt-4o-mini \
333
+ --llm-api-key-file ~/keys/openai_api_key.txt
260
334
  """
261
335
  e = env_of(env)
262
336
  safety.confirm_prod_strict("synthetic generation", env)
263
-
264
- # A comma list of per-study counts must line up with the studies given.
265
- if num_records and "," in num_records:
266
- n_counts = len(num_records.split(","))
267
- n_studies = len(studies.split(","))
268
- if n_counts != n_studies:
269
- typer.secho(
270
- f"--num-records has {n_counts} values but {n_studies} studies "
271
- f"were given (pass one count, or one per study).",
272
- fg=typer.colors.RED,
273
- err=True,
274
- )
275
- raise typer.Exit(1)
337
+ _check_per_study_counts(studies, num_records)
276
338
 
277
339
  ver = version or e.dictionary_version
278
340
  schema_path = schema or str(SCHEMA_DIR / dictionary_filename(e, ver))
@@ -428,8 +428,24 @@ def script_env(e: EnvConfig, version: Optional[str] = None) -> Dict[str, str]:
428
428
  scripts compose a URL keeps Python the single source of truth: the scripts
429
429
  read ``G3DT_DICT_URL``/``G3DT_DICT_FILENAME`` and never build either
430
430
  themselves, so there is only one implementation to keep correct.
431
+
432
+ The interpreter's own bin directory is prepended to PATH: console scripts
433
+ installed next to g3dt — gen3-metadata-simulator via
434
+ ``g3dt synth install-simulator`` — land there, but a pipx install exposes
435
+ only g3dt's own entry points on the caller's PATH, so without this the
436
+ wrapped scripts' ``command -v gen3-metadata-simulator`` check fails even
437
+ though the tool is installed.
431
438
  """
439
+ import sys
440
+
432
441
  env = dict(os.environ)
442
+ # No .resolve(): a venv's python is a symlink to the base interpreter, and
443
+ # resolving it would point at the base install's bin instead of the venv
444
+ # bin where console scripts are actually created.
445
+ venv_bin = str(Path(sys.executable).parent)
446
+ path_entries = env.get("PATH", "").split(os.pathsep)
447
+ if venv_bin not in path_entries:
448
+ env["PATH"] = venv_bin + os.pathsep + env.get("PATH", "")
433
449
  values = {
434
450
  "G3DT_ENV": e.name,
435
451
  "G3DT_REGION": e.region,
@@ -43,10 +43,24 @@ ARGO_SCRIPT_DIR="${SERVICE_DIR}/k8s_ops"
43
43
  SCHEMA_DIR="${G3DT_SCHEMA_DIR:-$HOME/.g3dt/schemas}"
44
44
  SYNTH_BASE="${G3DT_SYNTH_DIR:-$HOME/.g3dt/synth_metadata}"
45
45
 
46
+ # Batch inputs (set by `g3dt synth deploy --studies/--num-records/--prev-version`;
47
+ # the fallbacks are the original ACDC demo values, kept for continuity).
48
+ STUDIES="${G3DT_SYNTH_STUDIES:-AusDiab_Simulated,Baker-Biobank_Simulated,BioHeart-CT_Simulated,CAUGHT-CAD_Simulated}"
49
+ if [ -n "${G3DT_SYNTH_NUM_RECORDS:-}" ]; then
50
+ NUM_RECORDS="${G3DT_SYNTH_NUM_RECORDS}"
51
+ elif [ -n "${G3DT_SYNTH_STUDIES:-}" ]; then
52
+ NUM_RECORDS="30" # custom studies, no counts given: 30 per study
53
+ else
54
+ NUM_RECORDS="30,60,20,55" # the classic per-study counts for the demo set
55
+ fi
56
+ PREV_VERSION="${G3DT_SYNTH_PREV_VERSION:-v1.0.0}"
57
+
46
58
  # Derived variables
47
- PREV_VERSION="v1.0.0"
48
59
  SYNTH_META_DIR="${SYNTH_BASE}/${VERSION}/"
49
- DATA_IMPORT_ORDER_FILE="${SYNTH_BASE}/${PREV_VERSION}/AusDiab_Simulated/DataImportOrder.txt"
60
+ # The delete step walks the previous batch's import order; every study in a
61
+ # batch shares one, so the first study's copy serves.
62
+ FIRST_STUDY="${STUDIES%%,*}"
63
+ DATA_IMPORT_ORDER_FILE="${SYNTH_BASE}/${PREV_VERSION}/${FIRST_STUDY}/DataImportOrder.txt"
50
64
 
51
65
  # Never export an empty AWS_PROFILE (empty means ambient credentials).
52
66
  if [ -n "${G3DT_AWS_PROFILE:-}" ]; then
@@ -87,9 +101,9 @@ bash "${ARGO_SCRIPT_DIR}/argocd_restart_schema.sh" \
87
101
  -a "${APP_NAME}" \
88
102
  -n "${NAMESPACE}"
89
103
 
90
- echo "==== [4] Deleting old synthetic data for version ${PREV_VERSION} ===="
104
+ echo "==== [4] Deleting old synthetic data (${STUDIES}) for version ${PREV_VERSION} ===="
91
105
  DELETE_SYNTH_ARGS=(
92
- -p "AusDiab_Simulated,EDCAD-PMS_Simulated,PREDICT_Simulated,Baker-Biobank_Simulated,CAUGHT-CAD_Simulated,BioHeart-CT_Simulated"
106
+ -p "${STUDIES}"
93
107
  -s "${AWS_SECRET_NAME}"
94
108
  -i "${DATA_IMPORT_ORDER_FILE}"
95
109
  )
@@ -98,12 +112,13 @@ if [ -n "${G3DT_AWS_PROFILE:-}" ]; then
98
112
  fi
99
113
  python3 "${SERVICE_DIR}/synthetic_data/delete_synth_metadata_sheepdog.py" "${DELETE_SYNTH_ARGS[@]}"
100
114
 
101
- echo "==== [5] Generating new synthetic data for version ${VERSION} (LLM-realistic) ===="
115
+ echo "==== [5] Generating new synthetic data (${STUDIES}) for version ${VERSION} (LLM-realistic) ===="
102
116
  bash "${SERVICE_DIR}/synthetic_data/generate_synth_metadata.sh" \
103
117
  --schema "${SCHEMA_DIR}/${DICT_FILENAME}" \
104
118
  --version "${VERSION}" \
105
119
  --provider llm \
106
- --num-records "30,60,20,55" \
120
+ --studies "${STUDIES}" \
121
+ --num-records "${NUM_RECORDS}" \
107
122
  --output-root "${SYNTH_BASE}"
108
123
 
109
124
  echo "==== [6] Uploading new synthetic data for version ${VERSION} ===="