@aws-cdk/aws-glue-alpha 2.267.0-alpha.0 → 2.269.0-alpha.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.jsii +1691 -1261
- package/.jsii.tabl.json.gz +0 -0
- package/.warnings.jsii.js +1 -79
- package/README.md +106 -33
- package/adr/job-arguments.md +212 -0
- package/lib/catalog.d.ts +2 -2
- package/lib/catalog.js +4 -4
- package/lib/code.js +3 -3
- package/lib/connection.d.ts +44 -29
- package/lib/connection.js +66 -32
- package/lib/constants.d.ts +17 -0
- package/lib/constants.js +20 -2
- package/lib/data-format.js +5 -5
- package/lib/data-quality-ruleset.d.ts +20 -9
- package/lib/data-quality-ruleset.js +33 -6
- package/lib/database.d.ts +3 -1
- package/lib/database.js +8 -2
- package/lib/external-table.js +1 -1
- package/lib/jobs/job.d.ts +91 -10
- package/lib/jobs/job.js +115 -38
- package/lib/jobs/pyspark-etl-job.d.ts +2 -4
- package/lib/jobs/pyspark-etl-job.js +10 -15
- package/lib/jobs/pyspark-flex-etl-job.d.ts +2 -4
- package/lib/jobs/pyspark-flex-etl-job.js +10 -15
- package/lib/jobs/pyspark-streaming-job.d.ts +2 -4
- package/lib/jobs/pyspark-streaming-job.js +10 -15
- package/lib/jobs/python-shell-job.d.ts +20 -7
- package/lib/jobs/python-shell-job.js +30 -33
- package/lib/jobs/ray-job.js +8 -13
- package/lib/jobs/scala-spark-etl-job.d.ts +4 -5
- package/lib/jobs/scala-spark-etl-job.js +13 -16
- package/lib/jobs/scala-spark-flex-etl-job.d.ts +12 -23
- package/lib/jobs/scala-spark-flex-etl-job.js +20 -23
- package/lib/jobs/scala-spark-streaming-job.d.ts +4 -5
- package/lib/jobs/scala-spark-streaming-job.js +13 -16
- package/lib/jobs/spark-job.d.ts +9 -7
- package/lib/jobs/spark-job.js +23 -32
- package/lib/partition-projection.d.ts +36 -47
- package/lib/partition-projection.js +22 -40
- package/lib/s3-table.d.ts +7 -4
- package/lib/s3-table.js +12 -9
- package/lib/schema.js +2 -2
- package/lib/security-configuration.js +4 -4
- package/lib/storage-parameter.js +1 -1
- package/lib/table-base.js +1 -1
- package/lib/triggers/trigger-options.d.ts +103 -43
- package/lib/triggers/trigger-options.js +202 -3
- package/lib/triggers/workflow.d.ts +31 -47
- package/lib/triggers/workflow.js +27 -142
- package/package.json +7 -7
package/.jsii.tabl.json.gz
CHANGED
|
Binary file
|
package/.warnings.jsii.js
CHANGED
|
@@ -345,72 +345,7 @@ const VALIDATORS = { _aws_cdk_aws_glue_alpha_CatalogEncryptionOptions: function
|
|
|
345
345
|
finally {
|
|
346
346
|
visitedObjects.delete(p);
|
|
347
347
|
}
|
|
348
|
-
},
|
|
349
|
-
if (p == null)
|
|
350
|
-
return;
|
|
351
|
-
visitedObjects.add(p);
|
|
352
|
-
try {
|
|
353
|
-
if (p.actions != null)
|
|
354
|
-
for (const o of p.actions)
|
|
355
|
-
if (!visitedObjects.has(o))
|
|
356
|
-
module.exports._aws_cdk_aws_glue_alpha_Action(o);
|
|
357
|
-
}
|
|
358
|
-
finally {
|
|
359
|
-
visitedObjects.delete(p);
|
|
360
|
-
}
|
|
361
|
-
}, _aws_cdk_aws_glue_alpha_OnDemandTriggerOptions: function _aws_cdk_aws_glue_alpha_OnDemandTriggerOptions(p) {
|
|
362
|
-
if (p == null)
|
|
363
|
-
return;
|
|
364
|
-
visitedObjects.add(p);
|
|
365
|
-
try {
|
|
366
|
-
if (p.actions != null)
|
|
367
|
-
for (const o of p.actions)
|
|
368
|
-
if (!visitedObjects.has(o))
|
|
369
|
-
module.exports._aws_cdk_aws_glue_alpha_Action(o);
|
|
370
|
-
}
|
|
371
|
-
finally {
|
|
372
|
-
visitedObjects.delete(p);
|
|
373
|
-
}
|
|
374
|
-
}, _aws_cdk_aws_glue_alpha_DailyScheduleTriggerOptions: function _aws_cdk_aws_glue_alpha_DailyScheduleTriggerOptions(p) {
|
|
375
|
-
if (p == null)
|
|
376
|
-
return;
|
|
377
|
-
visitedObjects.add(p);
|
|
378
|
-
try {
|
|
379
|
-
if (p.actions != null)
|
|
380
|
-
for (const o of p.actions)
|
|
381
|
-
if (!visitedObjects.has(o))
|
|
382
|
-
module.exports._aws_cdk_aws_glue_alpha_Action(o);
|
|
383
|
-
}
|
|
384
|
-
finally {
|
|
385
|
-
visitedObjects.delete(p);
|
|
386
|
-
}
|
|
387
|
-
}, _aws_cdk_aws_glue_alpha_WeeklyScheduleTriggerOptions: function _aws_cdk_aws_glue_alpha_WeeklyScheduleTriggerOptions(p) {
|
|
388
|
-
if (p == null)
|
|
389
|
-
return;
|
|
390
|
-
visitedObjects.add(p);
|
|
391
|
-
try {
|
|
392
|
-
if (p.actions != null)
|
|
393
|
-
for (const o of p.actions)
|
|
394
|
-
if (!visitedObjects.has(o))
|
|
395
|
-
module.exports._aws_cdk_aws_glue_alpha_Action(o);
|
|
396
|
-
}
|
|
397
|
-
finally {
|
|
398
|
-
visitedObjects.delete(p);
|
|
399
|
-
}
|
|
400
|
-
}, _aws_cdk_aws_glue_alpha_CustomScheduledTriggerOptions: function _aws_cdk_aws_glue_alpha_CustomScheduledTriggerOptions(p) {
|
|
401
|
-
if (p == null)
|
|
402
|
-
return;
|
|
403
|
-
visitedObjects.add(p);
|
|
404
|
-
try {
|
|
405
|
-
if (p.actions != null)
|
|
406
|
-
for (const o of p.actions)
|
|
407
|
-
if (!visitedObjects.has(o))
|
|
408
|
-
module.exports._aws_cdk_aws_glue_alpha_Action(o);
|
|
409
|
-
}
|
|
410
|
-
finally {
|
|
411
|
-
visitedObjects.delete(p);
|
|
412
|
-
}
|
|
413
|
-
}, _aws_cdk_aws_glue_alpha_NotifyEventTriggerOptions: function _aws_cdk_aws_glue_alpha_NotifyEventTriggerOptions(p) {
|
|
348
|
+
}, _aws_cdk_aws_glue_alpha_EventTriggerOptions: function _aws_cdk_aws_glue_alpha_EventTriggerOptions(p) {
|
|
414
349
|
if (p == null)
|
|
415
350
|
return;
|
|
416
351
|
visitedObjects.add(p);
|
|
@@ -425,19 +360,6 @@ const VALIDATORS = { _aws_cdk_aws_glue_alpha_CatalogEncryptionOptions: function
|
|
|
425
360
|
finally {
|
|
426
361
|
visitedObjects.delete(p);
|
|
427
362
|
}
|
|
428
|
-
}, _aws_cdk_aws_glue_alpha_ConditionalTriggerOptions: function _aws_cdk_aws_glue_alpha_ConditionalTriggerOptions(p) {
|
|
429
|
-
if (p == null)
|
|
430
|
-
return;
|
|
431
|
-
visitedObjects.add(p);
|
|
432
|
-
try {
|
|
433
|
-
if (p.actions != null)
|
|
434
|
-
for (const o of p.actions)
|
|
435
|
-
if (!visitedObjects.has(o))
|
|
436
|
-
module.exports._aws_cdk_aws_glue_alpha_Action(o);
|
|
437
|
-
}
|
|
438
|
-
finally {
|
|
439
|
-
visitedObjects.delete(p);
|
|
440
|
-
}
|
|
441
363
|
} };
|
|
442
364
|
function print(name, deprecationMessage) {
|
|
443
365
|
const deprecated = process.env.JSII_DEPRECATED;
|
package/README.md
CHANGED
|
@@ -71,7 +71,7 @@ for more granular details.
|
|
|
71
71
|
#### ETL Jobs
|
|
72
72
|
|
|
73
73
|
ETL jobs support pySpark and Scala languages, for which there are separate but
|
|
74
|
-
similar constructors. ETL jobs default to the
|
|
74
|
+
similar constructors. ETL jobs default to the G1 worker type, but you can
|
|
75
75
|
override this default with other supported worker type values (G1, G2, G4
|
|
76
76
|
and G8). ETL jobs defaults to Glue version 4.0, which you can override to 3.0.
|
|
77
77
|
The following ETL features are enabled by default:
|
|
@@ -81,6 +81,17 @@ The Spark UI (`—enable-spark-ui`) is off by default; enable it by setting the
|
|
|
81
81
|
You can find more details about version, worker type and other features in
|
|
82
82
|
[Glue's public documentation](https://docs.aws.amazon.com/glue/latest/dg/aws-glue-api-jobs-job.html).
|
|
83
83
|
|
|
84
|
+
> **Note on continuous logging and encryption:** Because continuous logging is
|
|
85
|
+
> enabled by default, job driver and executor stdout/stderr are streamed to
|
|
86
|
+
> CloudWatch. Unless you attach a [`SecurityConfiguration`](#securityconfiguration)
|
|
87
|
+
> with `cloudWatchEncryption`, these logs are written to the account-shared,
|
|
88
|
+
> default Glue log group (`/aws-glue/jobs/logs-v2/`), which is **not** encrypted
|
|
89
|
+
> with a customer-managed key. Since job logs can contain sensitive runtime data
|
|
90
|
+
> (SQL statements, row values, error stack traces), attach a `SecurityConfiguration`
|
|
91
|
+
> with `cloudWatchEncryption` for regulated workloads. The construct emits a
|
|
92
|
+
> synthesis-time warning when continuous logging is on and no `SecurityConfiguration`
|
|
93
|
+
> is attached.
|
|
94
|
+
|
|
84
95
|
Reference the pyspark-etl-jobs.test.ts and scalaspark-etl-jobs.test.ts unit tests
|
|
85
96
|
for examples of required-only and optional job parameters when creating these
|
|
86
97
|
types of jobs.
|
|
@@ -140,7 +151,7 @@ Streaming jobs are similar to ETL jobs, except that they perform ETL on data
|
|
|
140
151
|
streams using the Apache Spark Structured Streaming framework. Some Spark
|
|
141
152
|
job features are not available to Streaming ETL jobs. They support Scala
|
|
142
153
|
and pySpark languages. PySpark streaming jobs run on Python 3. It
|
|
143
|
-
defaults to the
|
|
154
|
+
defaults to the G1 worker type and Glue 4.0, both of which you can override.
|
|
144
155
|
The following best practice features are enabled by default:
|
|
145
156
|
`—enable-metrics, —enable-continuous-cloudwatch-log`.
|
|
146
157
|
The Spark UI (`—enable-spark-ui`) is off by default; enable it by setting the
|
|
@@ -197,7 +208,7 @@ new glue.PySparkStreamingJob(stack, 'PySparkStreamingJob', {
|
|
|
197
208
|
|
|
198
209
|
The flexible execution class is appropriate for non-urgent jobs such as
|
|
199
210
|
pre-production jobs, testing, and one-time data loads. Flexible jobs default
|
|
200
|
-
to Glue version 5.0 and worker type `
|
|
211
|
+
to Glue version 5.0 and worker type `G_1X`. The following best practice
|
|
201
212
|
features are enabled by default:
|
|
202
213
|
`—enable-metrics, —enable-continuous-cloudwatch-log`
|
|
203
214
|
The Spark UI (`—enable-spark-ui`) is off by default; enable it by setting the
|
|
@@ -256,8 +267,9 @@ Python shell jobs support a Python version that depends on the AWS Glue
|
|
|
256
267
|
version you use. These can be used to schedule and run tasks that don't
|
|
257
268
|
require an Apache Spark environment. Python shell jobs default to
|
|
258
269
|
Python 3.9 and a MaxCapacity of `0.0625`. Python 3.9 supports pre-loaded
|
|
259
|
-
analytics libraries
|
|
260
|
-
|
|
270
|
+
analytics libraries, enabled by default (`librarySet: glue.LibrarySet.ANALYTICS`).
|
|
271
|
+
Set `librarySet: glue.LibrarySet.NONE` when your libraries are custom or
|
|
272
|
+
conflict with the pre-installed ones.
|
|
261
273
|
|
|
262
274
|
Reference the pyspark-shell-job.test.ts unit tests for examples of
|
|
263
275
|
required-only and optional job parameters when creating these types of jobs.
|
|
@@ -342,6 +354,51 @@ new glue.PySparkEtlJob(stack, 'SelectiveJob', {
|
|
|
342
354
|
|
|
343
355
|
This feature is available for all Spark job types (ETL, Streaming, Flex).
|
|
344
356
|
|
|
357
|
+
### Job Arguments
|
|
358
|
+
|
|
359
|
+
Glue jobs are configured through a map of name-value arguments (`DefaultArguments`). This construct
|
|
360
|
+
manages several of these arguments on your behalf and exposes each one through a dedicated,
|
|
361
|
+
strongly-typed prop:
|
|
362
|
+
|
|
363
|
+
| Managed argument(s) | Prop |
|
|
364
|
+
|--------------------------------------------------------------------------|-----------------------------------------------------------------|
|
|
365
|
+
| `--enable-continuous-cloudwatch-log`, `--continuous-log-*` | `continuousLogging` |
|
|
366
|
+
| `--enable-metrics` | `enableMetrics` |
|
|
367
|
+
| `--enable-observability-metrics` | `enableObservabilityMetrics` |
|
|
368
|
+
| `--enable-spark-ui`, `--spark-event-logs-path` | `sparkUI` |
|
|
369
|
+
| `--job-language`, `--class` | job class / `className` |
|
|
370
|
+
| `--extra-jars`, `--user-jars-first`, `--extra-py-files`, `--extra-files` | `extraJars`, `extraJarsFirst`, `extraPythonFiles`, `extraFiles` |
|
|
371
|
+
| `library-set` | `librarySet` (Python Shell) |
|
|
372
|
+
|
|
373
|
+
The `defaultArguments` prop is the escape hatch for arguments this construct does **not** model.
|
|
374
|
+
Use it for any argument without a dedicated prop:
|
|
375
|
+
|
|
376
|
+
```ts
|
|
377
|
+
import * as cdk from 'aws-cdk-lib';
|
|
378
|
+
import * as iam from 'aws-cdk-lib/aws-iam';
|
|
379
|
+
declare const stack: cdk.Stack;
|
|
380
|
+
declare const role: iam.IRole;
|
|
381
|
+
declare const script: glue.Code;
|
|
382
|
+
|
|
383
|
+
new glue.PySparkEtlJob(stack, 'PySparkETLJob', {
|
|
384
|
+
role,
|
|
385
|
+
script,
|
|
386
|
+
defaultArguments: {
|
|
387
|
+
// an argument this construct does not manage
|
|
388
|
+
'--enable-glue-datacatalog': 'true',
|
|
389
|
+
},
|
|
390
|
+
});
|
|
391
|
+
```
|
|
392
|
+
|
|
393
|
+
To keep a single, unambiguous way to express each intent, setting a **construct-managed** argument
|
|
394
|
+
(any argument in the table above) or a **Glue-reserved** argument (`--debug`, `--mode`,
|
|
395
|
+
`--JOB_NAME`, `--endpoint`) through `defaultArguments` throws at synthesis time. This holds even
|
|
396
|
+
when the feature is turned off — for example, `enableMetrics: false` combined with
|
|
397
|
+
`defaultArguments: { '--enable-metrics': '' }` throws rather than silently re-enabling metrics.
|
|
398
|
+
Configure managed arguments through their dedicated prop instead — for example, use
|
|
399
|
+
`continuousLogging: { enabled: false }` rather than
|
|
400
|
+
`defaultArguments: { '--enable-continuous-cloudwatch-log': 'false' }`.
|
|
401
|
+
|
|
345
402
|
### Enable Job Run Queuing
|
|
346
403
|
|
|
347
404
|
AWS Glue job queuing monitors your account level quotas and limits. If quotas or limits are insufficient to start a Glue job run, AWS Glue will automatically queue the job and wait for limits to free up. Once limits become available, AWS Glue will retry the job run. Glue jobs will queue for limits like max concurrent job runs per account, max concurrent Data Processing Units (DPU), and resource unavailable due to IP address exhaustion in Amazon Virtual Private Cloud (Amazon VPC).
|
|
@@ -405,7 +462,7 @@ const job = new glue.PySparkEtlJob(stack, 'Job', { role, script });
|
|
|
405
462
|
// Create a workflow and add a trigger that runs the job
|
|
406
463
|
const workflow = new glue.Workflow(stack, 'Workflow');
|
|
407
464
|
workflow.addOnDemandTrigger('OnDemandTrigger', {
|
|
408
|
-
actions: [
|
|
465
|
+
actions: [glue.Action.job(job)],
|
|
409
466
|
});
|
|
410
467
|
```
|
|
411
468
|
|
|
@@ -418,21 +475,35 @@ actions list using the job or crawler objects using conditional types.
|
|
|
418
475
|
|
|
419
476
|
#### **2. Scheduled Triggers**
|
|
420
477
|
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
without
|
|
426
|
-
|
|
427
|
-
|
|
478
|
+
Use `addScheduledTrigger` with a `TriggerSchedule` to fire on a cron schedule.
|
|
479
|
+
`TriggerSchedule.daily()` and `TriggerSchedule.weekly()` are convenience
|
|
480
|
+
factories; `TriggerSchedule.cron(...)` lets you build any schedule from the
|
|
481
|
+
[existing event Schedule class](https://docs.aws.amazon.com/cdk/api/v2/docs/aws-cdk-lib.aws_events.Schedule.html)
|
|
482
|
+
without writing raw cron expressions. The L2 extracts the expression that Glue
|
|
483
|
+
requires from the `TriggerSchedule`.
|
|
484
|
+
|
|
485
|
+
```ts
|
|
486
|
+
import * as cdk from 'aws-cdk-lib';
|
|
487
|
+
import * as iam from 'aws-cdk-lib/aws-iam';
|
|
488
|
+
declare const stack: cdk.Stack;
|
|
489
|
+
declare const role: iam.IRole;
|
|
490
|
+
declare const script: glue.Code;
|
|
491
|
+
const job = new glue.PySparkEtlJob(stack, 'Job', { role, script });
|
|
492
|
+
const workflow = new glue.Workflow(stack, 'Workflow');
|
|
493
|
+
|
|
494
|
+
workflow.addScheduledTrigger('WeeklyTrigger', {
|
|
495
|
+
actions: [glue.Action.job(job)],
|
|
496
|
+
schedule: glue.TriggerSchedule.weekly(),
|
|
497
|
+
});
|
|
498
|
+
```
|
|
428
499
|
|
|
429
|
-
#### **3.
|
|
500
|
+
#### **3. Event Triggers**
|
|
430
501
|
|
|
431
|
-
|
|
432
|
-
For batching triggers, you must specify `
|
|
433
|
-
triggers, `
|
|
434
|
-
defaults to 900 seconds, but you can override the window to align with
|
|
435
|
-
|
|
502
|
+
Use `addEventTrigger` for EventBridge event-based triggers. There are two types:
|
|
503
|
+
batching and non-batching. For batching triggers, you must specify `batchSize`.
|
|
504
|
+
For non-batching triggers, `batchSize` defaults to 1. For both, `batchWindow`
|
|
505
|
+
defaults to 900 seconds, but you can override the window to align with your
|
|
506
|
+
workload's requirements.
|
|
436
507
|
|
|
437
508
|
#### **4. Conditional Triggers**
|
|
438
509
|
|
|
@@ -451,13 +522,14 @@ certain types of data stores.
|
|
|
451
522
|
|
|
452
523
|
* **Networking - the CDK determines the best fit subnet for Glue connection
|
|
453
524
|
configuration**
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
`vpcSubnets`
|
|
525
|
+
Configure VPC placement through the `network` property, built with
|
|
526
|
+
`ConnectionNetwork.subnet(subnet)` to pin a specific subnet, or
|
|
527
|
+
`ConnectionNetwork.vpc(vpc, vpcSubnets?)` to let the L2 select one via the
|
|
528
|
+
existing
|
|
457
529
|
[EC2 Subnet Selection](https://docs.aws.amazon.com/cdk/api/v2/python/aws_cdk.aws_ec2/SubnetSelection.html)
|
|
458
|
-
library
|
|
459
|
-
|
|
460
|
-
|
|
530
|
+
library. A Glue connection targets a single subnet, so the first subnet of
|
|
531
|
+
the selection is used. The two factories are mutually exclusive, so a subnet
|
|
532
|
+
and a VPC can never be combined.
|
|
461
533
|
|
|
462
534
|
Pin the connection to a specific subnet:
|
|
463
535
|
|
|
@@ -469,7 +541,7 @@ new glue.Connection(this, 'MyConnection', {
|
|
|
469
541
|
// The security groups granting AWS Glue inbound access to the data source within the VPC
|
|
470
542
|
securityGroups: [securityGroup],
|
|
471
543
|
// The VPC subnet which contains the data source
|
|
472
|
-
subnet,
|
|
544
|
+
network: glue.ConnectionNetwork.subnet(subnet),
|
|
473
545
|
});
|
|
474
546
|
```
|
|
475
547
|
|
|
@@ -481,9 +553,8 @@ declare const vpc: ec2.Vpc;
|
|
|
481
553
|
new glue.Connection(this, 'MyConnection', {
|
|
482
554
|
type: glue.ConnectionType.NETWORK,
|
|
483
555
|
securityGroups: [securityGroup],
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
vpcSubnets: { subnetType: ec2.SubnetType.PRIVATE_WITH_EGRESS },
|
|
556
|
+
// vpcSubnets is optional - defaults to private subnets
|
|
557
|
+
network: glue.ConnectionNetwork.vpc(vpc, { subnetType: ec2.SubnetType.PRIVATE_WITH_EGRESS }),
|
|
487
558
|
});
|
|
488
559
|
```
|
|
489
560
|
|
|
@@ -496,7 +567,7 @@ declare const db: rds.DatabaseCluster;
|
|
|
496
567
|
new glue.Connection(this, "RdsConnection", {
|
|
497
568
|
type: glue.ConnectionType.JDBC,
|
|
498
569
|
securityGroups: [securityGroup],
|
|
499
|
-
subnet,
|
|
570
|
+
network: glue.ConnectionNetwork.subnet(subnet),
|
|
500
571
|
secret: db.secret,
|
|
501
572
|
properties: {
|
|
502
573
|
JDBC_CONNECTION_URL: `jdbc:mysql://${db.clusterEndpoint.socketAddress}/databasename`,
|
|
@@ -945,8 +1016,9 @@ new glue.S3Table(this, 'MyTable', {
|
|
|
945
1016
|
min: '2020-01-01',
|
|
946
1017
|
max: '2023-12-31',
|
|
947
1018
|
format: 'yyyy-MM-dd',
|
|
948
|
-
interval
|
|
949
|
-
|
|
1019
|
+
// `step` bundles interval + unit (supply both or neither). Optional at day
|
|
1020
|
+
// precision or coarser; required when the format is sub-day (e.g. hours).
|
|
1021
|
+
step: { interval: 1, intervalUnit: glue.DateIntervalUnit.DAYS },
|
|
950
1022
|
}),
|
|
951
1023
|
},
|
|
952
1024
|
});
|
|
@@ -1098,10 +1170,11 @@ Data Quality Definition Language (DQDL) — that are evaluated against a table i
|
|
|
1098
1170
|
the Data Catalog.
|
|
1099
1171
|
|
|
1100
1172
|
```ts
|
|
1173
|
+
declare const database: glue.IDatabase;
|
|
1101
1174
|
new glue.DataQualityRuleset(this, 'MyRuleset', {
|
|
1102
1175
|
rulesetName: 'my_ruleset',
|
|
1103
1176
|
dqdl: glue.Dqdl.fromString('Rules = [ RowCount > 100, IsComplete "order_id" ]'),
|
|
1104
|
-
targetTable:
|
|
1177
|
+
targetTable: glue.DataQualityTargetTable.fromTableName(database, 'my_table'),
|
|
1105
1178
|
});
|
|
1106
1179
|
```
|
|
1107
1180
|
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
# Glue Job Arguments
|
|
2
|
+
|
|
3
|
+
## Status
|
|
4
|
+
|
|
5
|
+
accepted
|
|
6
|
+
|
|
7
|
+
## Context
|
|
8
|
+
|
|
9
|
+
Every Glue job resource (`AWS::Glue::Job`) accepts a `DefaultArguments` map — a
|
|
10
|
+
flat `string → string` dictionary of `--flag`/value pairs that Glue passes to the
|
|
11
|
+
job script on every run. Some of these arguments are ordinary user configuration
|
|
12
|
+
(`--additional-python-modules`, `--enable-glue-datacatalog`, `--TempDir`, …), but
|
|
13
|
+
others are the wire form of features the CDK L2 models with strongly-typed props.
|
|
14
|
+
|
|
15
|
+
The L2 job constructs therefore populate `DefaultArguments` from two sources:
|
|
16
|
+
|
|
17
|
+
1. **Construct-managed arguments** — derived by the construct from typed props or
|
|
18
|
+
from the job class itself. Examples:
|
|
19
|
+
- `continuousLogging` → `--enable-continuous-cloudwatch-log`,
|
|
20
|
+
`--continuous-log-logGroup`, `--continuous-log-logStreamPrefix`,
|
|
21
|
+
`--continuous-log-conversionPattern`, `--enable-continuous-log-filter`
|
|
22
|
+
- `enableMetrics` → `--enable-metrics`
|
|
23
|
+
- `enableObservabilityMetrics` → `--enable-observability-metrics`
|
|
24
|
+
- `sparkUI` → `--enable-spark-ui`, `--spark-event-logs-path`
|
|
25
|
+
- `extraJars` / `extraJarsFirst` / `extraPythonFiles` / `extraFiles` →
|
|
26
|
+
`--extra-jars`, `--user-jars-first`, `--extra-py-files`, `--extra-files`
|
|
27
|
+
- `className` → `--class` (Scala only)
|
|
28
|
+
- the job language itself → `--job-language`
|
|
29
|
+
- `librarySet` → `library-set` (Python Shell only)
|
|
30
|
+
2. **`defaultArguments`** — the untyped escape-hatch map the user supplies directly,
|
|
31
|
+
for arguments the L2 does *not* model.
|
|
32
|
+
|
|
33
|
+
The two sources can collide. Before this decision, the collision was resolved
|
|
34
|
+
silently and inconsistently across job types:
|
|
35
|
+
|
|
36
|
+
- `SparkJob` / `PythonShellJob` merged as `{ ...managed, ...userDefaultArguments }`,
|
|
37
|
+
so the user value won — a user could pass
|
|
38
|
+
`defaultArguments: { '--enable-continuous-cloudwatch-log': 'false' }` and silently
|
|
39
|
+
turn off a secure default.
|
|
40
|
+
- `RayJob` merged the other way, so the construct value won — a user's
|
|
41
|
+
`defaultArguments` entry for a managed key was silently dropped.
|
|
42
|
+
|
|
43
|
+
Both behaviors are footguns: one weakens the construct's secure/observable defaults
|
|
44
|
+
without warning, the other ignores explicit user input without warning. Because Glue
|
|
45
|
+
enables continuous CloudWatch logging by default and that data can contain sensitive
|
|
46
|
+
runtime values (SQL, row data, stack traces), the "user silently wins" case is also a
|
|
47
|
+
security concern.
|
|
48
|
+
|
|
49
|
+
Separately, Glue itself reserves a handful of argument keys for its own internal use
|
|
50
|
+
(`--debug`, `--mode`, `--JOB_NAME`, `--endpoint`). These are never valid user input on
|
|
51
|
+
any job type.
|
|
52
|
+
|
|
53
|
+
## Constraints
|
|
54
|
+
|
|
55
|
+
- `DefaultArguments` is a single flat map on the L1; there is no separate channel to
|
|
56
|
+
distinguish "managed" from "user" keys once they are merged. Whatever the L2 does,
|
|
57
|
+
it must produce one merged map.
|
|
58
|
+
- Which keys are managed varies by job type: `--class` exists only for Scala jobs,
|
|
59
|
+
`library-set` only for Python Shell, `--enable-spark-ui` only for Spark, and so on.
|
|
60
|
+
A single global list would either over-reject (block a key that is a legitimate
|
|
61
|
+
escape hatch for a job type that doesn't manage it — e.g. `--extra-py-files` on a
|
|
62
|
+
job with no `extraPythonFiles` prop) or under-reject.
|
|
63
|
+
- Argument keys can be tokens (e.g. produced by `CfnJson`) that only resolve at
|
|
64
|
+
deploy time. String comparison cannot see through them at synthesis.
|
|
65
|
+
- The set of managed keys must not be tied to the set the construct *happens to emit*
|
|
66
|
+
for a given configuration: a prop that turns a feature off (`enableMetrics:
|
|
67
|
+
false`) emits nothing, but the key is still construct-managed and must stay reserved.
|
|
68
|
+
|
|
69
|
+
## Decision
|
|
70
|
+
|
|
71
|
+
**A managed argument has exactly one way to be configured: its typed prop.** Passing a
|
|
72
|
+
construct-managed or Glue-reserved key through `defaultArguments` throws a
|
|
73
|
+
`ValidationError` at synthesis time rather than silently winning or being dropped.
|
|
74
|
+
`defaultArguments` remains the escape hatch for every argument the L2 does not model.
|
|
75
|
+
|
|
76
|
+
### Data flow
|
|
77
|
+
|
|
78
|
+
There is a single sink for every construct-managed argument — the base-class method:
|
|
79
|
+
|
|
80
|
+
```ts
|
|
81
|
+
protected setManagedArgument(key: string, value?: string): void
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
It records `key` in the reserved set and, when `value !== undefined`, emits it. A
|
|
85
|
+
subclass calls it once per managed key, passing `undefined` when the feature is off or
|
|
86
|
+
unset — the key is reserved either way. Subclasses do not build local argument maps, so
|
|
87
|
+
this is the *only* way to emit a managed argument: declaration and emission happen in the
|
|
88
|
+
same call, and the reserved set therefore cannot drift from what is emitted.
|
|
89
|
+
|
|
90
|
+
Each job subclass, in its constructor:
|
|
91
|
+
|
|
92
|
+
1. Registers its managed arguments through `setManagedArgument` — directly, or through
|
|
93
|
+
the shared helpers `setupContinuousLogging` (all job types),
|
|
94
|
+
`nonExecutableCommonArguments` and `setupExtraCodeArguments` (Spark), and its own
|
|
95
|
+
`executableArguments` (`--job-language`; plus `--class` for Scala, `library-set` for
|
|
96
|
+
Python Shell).
|
|
97
|
+
2. Calls the base-class method:
|
|
98
|
+
|
|
99
|
+
```ts
|
|
100
|
+
protected mergeDefaultArguments(
|
|
101
|
+
defaultArguments?: { [key: string]: string },
|
|
102
|
+
): { [key: string]: string }
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
which validates the user-supplied `defaultArguments` against the accumulated reserved
|
|
106
|
+
set and returns the merged map, passed as `DefaultArguments` on the `CfnJob`.
|
|
107
|
+
|
|
108
|
+
`setManagedArgument` declares managed keys; `mergeDefaultArguments` validates and merges
|
|
109
|
+
them. Each is the sole choke point for its job.
|
|
110
|
+
|
|
111
|
+
Which keys a job type reserves falls out of which `setManagedArgument` calls its
|
|
112
|
+
constructor makes: only Scala jobs register `--class`, only Python Shell registers
|
|
113
|
+
`library-set`, only Spark registers `--enable-spark-ui`, and so on. Per-job-type scoping
|
|
114
|
+
is automatic — there is no separate list to maintain per type.
|
|
115
|
+
|
|
116
|
+
### The reserved set
|
|
117
|
+
|
|
118
|
+
The keys a user may not set through `defaultArguments` are the union of:
|
|
119
|
+
|
|
120
|
+
- **`GLUE_RESERVED_ARGUMENTS`** — `--debug`, `--mode`, `--JOB_NAME`, `--endpoint`. Owned
|
|
121
|
+
by the Glue service, reserved on every job type. This is the one static list, because
|
|
122
|
+
it is external to the constructs — nothing derives it from a prop.
|
|
123
|
+
- **`_managedArgumentKeys`** — every key that any `setManagedArgument` call registered on
|
|
124
|
+
this instance, whether or not a value was emitted for it.
|
|
125
|
+
|
|
126
|
+
### Validation rules (per user-supplied key)
|
|
127
|
+
|
|
128
|
+
For each key in `defaultArguments`:
|
|
129
|
+
|
|
130
|
+
1. If the key is an unresolved token, the conflict check is skipped (equality is
|
|
131
|
+
unknowable at synth time) and a warning
|
|
132
|
+
(`@aws-cdk/aws-glue-alpha:tokenJobArgumentKey`) is emitted. If it resolves to a
|
|
133
|
+
managed key at deploy time, the construct-managed value wins (see merge order).
|
|
134
|
+
2. If the key is in **`GLUE_RESERVED_ARGUMENTS`** → throw. Glue-reserved keys are
|
|
135
|
+
never emitted by the construct, so there is no construct value to reconcile against.
|
|
136
|
+
3. If the key is in the reserved set (`_managedArgumentKeys`):
|
|
137
|
+
- If the construct actually emitted a value for that key
|
|
138
|
+
(`Object.hasOwn(_managedArguments, key)`) and the supplied value is identical →
|
|
139
|
+
allowed. Passing the same value the construct would produce is not
|
|
140
|
+
contradictory; autocorrecting config is preferred over an error.
|
|
141
|
+
- Otherwise (different value, or the construct emitted nothing because the feature
|
|
142
|
+
is off) → throw.
|
|
143
|
+
4. Otherwise, the key is genuinely custom → allowed, flows through untouched.
|
|
144
|
+
|
|
145
|
+
`Object.hasOwn` is used deliberately instead of the `in` operator so that inherited
|
|
146
|
+
`Object.prototype` members (`toString`, `constructor`, `hasOwnProperty`, …) supplied as
|
|
147
|
+
argument keys are treated as ordinary custom keys rather than falsely matching a
|
|
148
|
+
managed key.
|
|
149
|
+
|
|
150
|
+
### Merge order
|
|
151
|
+
|
|
152
|
+
After validation, the result is:
|
|
153
|
+
|
|
154
|
+
```ts
|
|
155
|
+
return { ...defaultArguments, ...this._managedArguments };
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
Managed arguments are spread last, so they win on any residual overlap. By this point
|
|
159
|
+
the only overlaps that can remain are (a) exact-value matches allowed by rule 3, which
|
|
160
|
+
are indistinguishable either way, and (b) token keys from rule 1, for which
|
|
161
|
+
managed-wins is the documented and warned-about behavior.
|
|
162
|
+
|
|
163
|
+
### Related synthesis-time warnings
|
|
164
|
+
|
|
165
|
+
Two other warnings live in the same flow:
|
|
166
|
+
|
|
167
|
+
- **`@aws-cdk/aws-glue-alpha:unencryptedContinuousLogging`** — continuous logging is on
|
|
168
|
+
(explicitly or by default) but no `SecurityConfiguration` is attached, so driver /
|
|
169
|
+
executor logs land in an unencrypted, account-shared CloudWatch log group. We only
|
|
170
|
+
warn when *no* security configuration is attached at all, because
|
|
171
|
+
`ISecurityConfiguration` exposes only the name and we cannot introspect whether it
|
|
172
|
+
actually configures `cloudWatchEncryption` (avoiding false positives).
|
|
173
|
+
- **`@aws-cdk/aws-glue-alpha:plaintextJobArgumentSecret`** — a `defaultArguments` key
|
|
174
|
+
looks like a credential and holds a plaintext literal. `DefaultArguments` is emitted
|
|
175
|
+
verbatim into the template; secrets belong in AWS Secrets Manager.
|
|
176
|
+
|
|
177
|
+
## Alternatives
|
|
178
|
+
|
|
179
|
+
### Invert precedence so the construct always wins
|
|
180
|
+
|
|
181
|
+
Merge as `{ ...userDefaultArguments, ...managed }` everywhere (which is what `RayJob`
|
|
182
|
+
already did). This is more secure than "user wins" but still silent: a user who
|
|
183
|
+
deliberately sets a managed key via `defaultArguments` has it dropped with no
|
|
184
|
+
indication. It also still leaves two channels for one setting. Rejected in favor of a
|
|
185
|
+
single, explicit way to express each intent.
|
|
186
|
+
|
|
187
|
+
### Derive the reserved set from the emitted arguments
|
|
188
|
+
|
|
189
|
+
Compute conflicts from the keys the construct actually emits. Attractive: adding a typed
|
|
190
|
+
prop reserves its key automatically, with no separate list to maintain. Rejected: it ties
|
|
191
|
+
the reserved set to the current configuration. A prop that turns a feature off emits no
|
|
192
|
+
key, so `defaultArguments` could silently re-enable it (`enableMetrics: false` +
|
|
193
|
+
`defaultArguments: { '--enable-metrics': '' }`). The reserved set must include keys the
|
|
194
|
+
construct manages even when it emits nothing for them. That is why `setManagedArgument`
|
|
195
|
+
records the key regardless of value.
|
|
196
|
+
|
|
197
|
+
### A per-job-type list of managed keys
|
|
198
|
+
|
|
199
|
+
Give each job type an explicit `string[]` of the keys it manages, passed to the
|
|
200
|
+
validation step alongside the emitted map. Correct, but it names every managed key in
|
|
201
|
+
two places — the emission site (inside an `enabled ? {...} : {}` expression) and the
|
|
202
|
+
list — which drift: adding a typed prop requires updating the list too, and a forgotten
|
|
203
|
+
entry silently reopens the re-enable bypass with no compile-time signal. The
|
|
204
|
+
`setManagedArgument` accumulator collapses the two into one call, so the key is named
|
|
205
|
+
once.
|
|
206
|
+
|
|
207
|
+
### One global reserved list on the base class
|
|
208
|
+
|
|
209
|
+
A single list of every managed key across all job types. Rejected because it
|
|
210
|
+
over-rejects: it would block, for example, `--extra-py-files` on a job type that has no
|
|
211
|
+
`extraPythonFiles` prop, removing a legitimate escape hatch with no typed replacement.
|
|
212
|
+
Managed keys must be scoped per job type.
|
package/lib/catalog.d.ts
CHANGED
|
@@ -44,7 +44,7 @@ export declare class DataCatalogEncryptionAtRest {
|
|
|
44
44
|
* @param key the KMS key to use. If omitted, an AWS-managed key is used and
|
|
45
45
|
* the key is not exposed as a grantable resource.
|
|
46
46
|
*/
|
|
47
|
-
static kms(key?: kms.
|
|
47
|
+
static kms(key?: kms.IKeyRef): DataCatalogEncryptionAtRest;
|
|
48
48
|
/**
|
|
49
49
|
* Encrypt the Data Catalog at rest with an AWS KMS key, accessed through a
|
|
50
50
|
* service role that AWS Glue assumes on your behalf.
|
|
@@ -56,7 +56,7 @@ export declare class DataCatalogEncryptionAtRest {
|
|
|
56
56
|
* @param key the KMS key to use. If omitted, an AWS-managed key is used and
|
|
57
57
|
* the key is not exposed as a grantable resource.
|
|
58
58
|
*/
|
|
59
|
-
static kmsWithServiceRole(role: iam.IRole, key?: kms.
|
|
59
|
+
static kmsWithServiceRole(role: iam.IRole, key?: kms.IKeyRef): DataCatalogEncryptionAtRest;
|
|
60
60
|
/**
|
|
61
61
|
* The encryption mode.
|
|
62
62
|
*/
|