@aws-cdk/aws-glue-alpha 2.268.0-alpha.0 → 2.269.0-alpha.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.jsii +1086 -878
- package/.jsii.tabl.json.gz +0 -0
- package/.warnings.jsii.js +1 -79
- package/README.md +101 -29
- package/adr/job-arguments.md +212 -0
- package/lib/catalog.js +3 -3
- package/lib/code.js +3 -3
- package/lib/connection.d.ts +44 -29
- package/lib/connection.js +66 -32
- package/lib/constants.d.ts +17 -0
- package/lib/constants.js +20 -2
- package/lib/data-format.js +5 -5
- package/lib/data-quality-ruleset.js +3 -3
- package/lib/database.js +1 -1
- package/lib/external-table.js +1 -1
- package/lib/jobs/job.d.ts +90 -9
- package/lib/jobs/job.js +115 -38
- package/lib/jobs/pyspark-etl-job.d.ts +1 -3
- package/lib/jobs/pyspark-etl-job.js +9 -14
- package/lib/jobs/pyspark-flex-etl-job.d.ts +1 -3
- package/lib/jobs/pyspark-flex-etl-job.js +9 -14
- package/lib/jobs/pyspark-streaming-job.d.ts +1 -3
- package/lib/jobs/pyspark-streaming-job.js +9 -14
- package/lib/jobs/python-shell-job.d.ts +20 -7
- package/lib/jobs/python-shell-job.js +30 -33
- package/lib/jobs/ray-job.js +8 -13
- package/lib/jobs/scala-spark-etl-job.d.ts +1 -3
- package/lib/jobs/scala-spark-etl-job.js +10 -14
- package/lib/jobs/scala-spark-flex-etl-job.d.ts +1 -3
- package/lib/jobs/scala-spark-flex-etl-job.js +10 -14
- package/lib/jobs/scala-spark-streaming-job.d.ts +1 -3
- package/lib/jobs/scala-spark-streaming-job.js +10 -14
- package/lib/jobs/spark-job.d.ts +9 -7
- package/lib/jobs/spark-job.js +23 -32
- package/lib/partition-projection.d.ts +36 -47
- package/lib/partition-projection.js +22 -40
- package/lib/s3-table.js +4 -4
- package/lib/schema.js +2 -2
- package/lib/security-configuration.js +4 -4
- package/lib/storage-parameter.js +1 -1
- package/lib/table-base.js +1 -1
- package/lib/triggers/trigger-options.d.ts +102 -42
- package/lib/triggers/trigger-options.js +202 -3
- package/lib/triggers/workflow.d.ts +31 -47
- package/lib/triggers/workflow.js +27 -142
- package/package.json +7 -7
package/.jsii.tabl.json.gz
CHANGED
|
Binary file
|
package/.warnings.jsii.js
CHANGED
|
@@ -345,72 +345,7 @@ const VALIDATORS = { _aws_cdk_aws_glue_alpha_CatalogEncryptionOptions: function
|
|
|
345
345
|
finally {
|
|
346
346
|
visitedObjects.delete(p);
|
|
347
347
|
}
|
|
348
|
-
},
|
|
349
|
-
if (p == null)
|
|
350
|
-
return;
|
|
351
|
-
visitedObjects.add(p);
|
|
352
|
-
try {
|
|
353
|
-
if (p.actions != null)
|
|
354
|
-
for (const o of p.actions)
|
|
355
|
-
if (!visitedObjects.has(o))
|
|
356
|
-
module.exports._aws_cdk_aws_glue_alpha_Action(o);
|
|
357
|
-
}
|
|
358
|
-
finally {
|
|
359
|
-
visitedObjects.delete(p);
|
|
360
|
-
}
|
|
361
|
-
}, _aws_cdk_aws_glue_alpha_OnDemandTriggerOptions: function _aws_cdk_aws_glue_alpha_OnDemandTriggerOptions(p) {
|
|
362
|
-
if (p == null)
|
|
363
|
-
return;
|
|
364
|
-
visitedObjects.add(p);
|
|
365
|
-
try {
|
|
366
|
-
if (p.actions != null)
|
|
367
|
-
for (const o of p.actions)
|
|
368
|
-
if (!visitedObjects.has(o))
|
|
369
|
-
module.exports._aws_cdk_aws_glue_alpha_Action(o);
|
|
370
|
-
}
|
|
371
|
-
finally {
|
|
372
|
-
visitedObjects.delete(p);
|
|
373
|
-
}
|
|
374
|
-
}, _aws_cdk_aws_glue_alpha_DailyScheduleTriggerOptions: function _aws_cdk_aws_glue_alpha_DailyScheduleTriggerOptions(p) {
|
|
375
|
-
if (p == null)
|
|
376
|
-
return;
|
|
377
|
-
visitedObjects.add(p);
|
|
378
|
-
try {
|
|
379
|
-
if (p.actions != null)
|
|
380
|
-
for (const o of p.actions)
|
|
381
|
-
if (!visitedObjects.has(o))
|
|
382
|
-
module.exports._aws_cdk_aws_glue_alpha_Action(o);
|
|
383
|
-
}
|
|
384
|
-
finally {
|
|
385
|
-
visitedObjects.delete(p);
|
|
386
|
-
}
|
|
387
|
-
}, _aws_cdk_aws_glue_alpha_WeeklyScheduleTriggerOptions: function _aws_cdk_aws_glue_alpha_WeeklyScheduleTriggerOptions(p) {
|
|
388
|
-
if (p == null)
|
|
389
|
-
return;
|
|
390
|
-
visitedObjects.add(p);
|
|
391
|
-
try {
|
|
392
|
-
if (p.actions != null)
|
|
393
|
-
for (const o of p.actions)
|
|
394
|
-
if (!visitedObjects.has(o))
|
|
395
|
-
module.exports._aws_cdk_aws_glue_alpha_Action(o);
|
|
396
|
-
}
|
|
397
|
-
finally {
|
|
398
|
-
visitedObjects.delete(p);
|
|
399
|
-
}
|
|
400
|
-
}, _aws_cdk_aws_glue_alpha_CustomScheduledTriggerOptions: function _aws_cdk_aws_glue_alpha_CustomScheduledTriggerOptions(p) {
|
|
401
|
-
if (p == null)
|
|
402
|
-
return;
|
|
403
|
-
visitedObjects.add(p);
|
|
404
|
-
try {
|
|
405
|
-
if (p.actions != null)
|
|
406
|
-
for (const o of p.actions)
|
|
407
|
-
if (!visitedObjects.has(o))
|
|
408
|
-
module.exports._aws_cdk_aws_glue_alpha_Action(o);
|
|
409
|
-
}
|
|
410
|
-
finally {
|
|
411
|
-
visitedObjects.delete(p);
|
|
412
|
-
}
|
|
413
|
-
}, _aws_cdk_aws_glue_alpha_NotifyEventTriggerOptions: function _aws_cdk_aws_glue_alpha_NotifyEventTriggerOptions(p) {
|
|
348
|
+
}, _aws_cdk_aws_glue_alpha_EventTriggerOptions: function _aws_cdk_aws_glue_alpha_EventTriggerOptions(p) {
|
|
414
349
|
if (p == null)
|
|
415
350
|
return;
|
|
416
351
|
visitedObjects.add(p);
|
|
@@ -425,19 +360,6 @@ const VALIDATORS = { _aws_cdk_aws_glue_alpha_CatalogEncryptionOptions: function
|
|
|
425
360
|
finally {
|
|
426
361
|
visitedObjects.delete(p);
|
|
427
362
|
}
|
|
428
|
-
}, _aws_cdk_aws_glue_alpha_ConditionalTriggerOptions: function _aws_cdk_aws_glue_alpha_ConditionalTriggerOptions(p) {
|
|
429
|
-
if (p == null)
|
|
430
|
-
return;
|
|
431
|
-
visitedObjects.add(p);
|
|
432
|
-
try {
|
|
433
|
-
if (p.actions != null)
|
|
434
|
-
for (const o of p.actions)
|
|
435
|
-
if (!visitedObjects.has(o))
|
|
436
|
-
module.exports._aws_cdk_aws_glue_alpha_Action(o);
|
|
437
|
-
}
|
|
438
|
-
finally {
|
|
439
|
-
visitedObjects.delete(p);
|
|
440
|
-
}
|
|
441
363
|
} };
|
|
442
364
|
function print(name, deprecationMessage) {
|
|
443
365
|
const deprecated = process.env.JSII_DEPRECATED;
|
package/README.md
CHANGED
|
@@ -81,6 +81,17 @@ The Spark UI (`—enable-spark-ui`) is off by default; enable it by setting the
|
|
|
81
81
|
You can find more details about version, worker type and other features in
|
|
82
82
|
[Glue's public documentation](https://docs.aws.amazon.com/glue/latest/dg/aws-glue-api-jobs-job.html).
|
|
83
83
|
|
|
84
|
+
> **Note on continuous logging and encryption:** Because continuous logging is
|
|
85
|
+
> enabled by default, job driver and executor stdout/stderr are streamed to
|
|
86
|
+
> CloudWatch. Unless you attach a [`SecurityConfiguration`](#securityconfiguration)
|
|
87
|
+
> with `cloudWatchEncryption`, these logs are written to the account-shared,
|
|
88
|
+
> default Glue log group (`/aws-glue/jobs/logs-v2/`), which is **not** encrypted
|
|
89
|
+
> with a customer-managed key. Since job logs can contain sensitive runtime data
|
|
90
|
+
> (SQL statements, row values, error stack traces), attach a `SecurityConfiguration`
|
|
91
|
+
> with `cloudWatchEncryption` for regulated workloads. The construct emits a
|
|
92
|
+
> synthesis-time warning when continuous logging is on and no `SecurityConfiguration`
|
|
93
|
+
> is attached.
|
|
94
|
+
|
|
84
95
|
Reference the pyspark-etl-jobs.test.ts and scalaspark-etl-jobs.test.ts unit tests
|
|
85
96
|
for examples of required-only and optional job parameters when creating these
|
|
86
97
|
types of jobs.
|
|
@@ -256,8 +267,9 @@ Python shell jobs support a Python version that depends on the AWS Glue
|
|
|
256
267
|
version you use. These can be used to schedule and run tasks that don't
|
|
257
268
|
require an Apache Spark environment. Python shell jobs default to
|
|
258
269
|
Python 3.9 and a MaxCapacity of `0.0625`. Python 3.9 supports pre-loaded
|
|
259
|
-
analytics libraries
|
|
260
|
-
|
|
270
|
+
analytics libraries, enabled by default (`librarySet: glue.LibrarySet.ANALYTICS`).
|
|
271
|
+
Set `librarySet: glue.LibrarySet.NONE` when your libraries are custom or
|
|
272
|
+
conflict with the pre-installed ones.
|
|
261
273
|
|
|
262
274
|
Reference the pyspark-shell-job.test.ts unit tests for examples of
|
|
263
275
|
required-only and optional job parameters when creating these types of jobs.
|
|
@@ -342,6 +354,51 @@ new glue.PySparkEtlJob(stack, 'SelectiveJob', {
|
|
|
342
354
|
|
|
343
355
|
This feature is available for all Spark job types (ETL, Streaming, Flex).
|
|
344
356
|
|
|
357
|
+
### Job Arguments
|
|
358
|
+
|
|
359
|
+
Glue jobs are configured through a map of name-value arguments (`DefaultArguments`). This construct
|
|
360
|
+
manages several of these arguments on your behalf and exposes each one through a dedicated,
|
|
361
|
+
strongly-typed prop:
|
|
362
|
+
|
|
363
|
+
| Managed argument(s) | Prop |
|
|
364
|
+
|--------------------------------------------------------------------------|-----------------------------------------------------------------|
|
|
365
|
+
| `--enable-continuous-cloudwatch-log`, `--continuous-log-*` | `continuousLogging` |
|
|
366
|
+
| `--enable-metrics` | `enableMetrics` |
|
|
367
|
+
| `--enable-observability-metrics` | `enableObservabilityMetrics` |
|
|
368
|
+
| `--enable-spark-ui`, `--spark-event-logs-path` | `sparkUI` |
|
|
369
|
+
| `--job-language`, `--class` | job class / `className` |
|
|
370
|
+
| `--extra-jars`, `--user-jars-first`, `--extra-py-files`, `--extra-files` | `extraJars`, `extraJarsFirst`, `extraPythonFiles`, `extraFiles` |
|
|
371
|
+
| `library-set` | `librarySet` (Python Shell) |
|
|
372
|
+
|
|
373
|
+
The `defaultArguments` prop is the escape hatch for arguments this construct does **not** model.
|
|
374
|
+
Use it for any argument without a dedicated prop:
|
|
375
|
+
|
|
376
|
+
```ts
|
|
377
|
+
import * as cdk from 'aws-cdk-lib';
|
|
378
|
+
import * as iam from 'aws-cdk-lib/aws-iam';
|
|
379
|
+
declare const stack: cdk.Stack;
|
|
380
|
+
declare const role: iam.IRole;
|
|
381
|
+
declare const script: glue.Code;
|
|
382
|
+
|
|
383
|
+
new glue.PySparkEtlJob(stack, 'PySparkETLJob', {
|
|
384
|
+
role,
|
|
385
|
+
script,
|
|
386
|
+
defaultArguments: {
|
|
387
|
+
// an argument this construct does not manage
|
|
388
|
+
'--enable-glue-datacatalog': 'true',
|
|
389
|
+
},
|
|
390
|
+
});
|
|
391
|
+
```
|
|
392
|
+
|
|
393
|
+
To keep a single, unambiguous way to express each intent, setting a **construct-managed** argument
|
|
394
|
+
(any argument in the table above) or a **Glue-reserved** argument (`--debug`, `--mode`,
|
|
395
|
+
`--JOB_NAME`, `--endpoint`) through `defaultArguments` throws at synthesis time. This holds even
|
|
396
|
+
when the feature is turned off — for example, `enableMetrics: false` combined with
|
|
397
|
+
`defaultArguments: { '--enable-metrics': '' }` throws rather than silently re-enabling metrics.
|
|
398
|
+
Configure managed arguments through their dedicated prop instead — for example, use
|
|
399
|
+
`continuousLogging: { enabled: false }` rather than
|
|
400
|
+
`defaultArguments: { '--enable-continuous-cloudwatch-log': 'false' }`.
|
|
401
|
+
|
|
345
402
|
### Enable Job Run Queuing
|
|
346
403
|
|
|
347
404
|
AWS Glue job queuing monitors your account level quotas and limits. If quotas or limits are insufficient to start a Glue job run, AWS Glue will automatically queue the job and wait for limits to free up. Once limits become available, AWS Glue will retry the job run. Glue jobs will queue for limits like max concurrent job runs per account, max concurrent Data Processing Units (DPU), and resource unavailable due to IP address exhaustion in Amazon Virtual Private Cloud (Amazon VPC).
|
|
@@ -405,7 +462,7 @@ const job = new glue.PySparkEtlJob(stack, 'Job', { role, script });
|
|
|
405
462
|
// Create a workflow and add a trigger that runs the job
|
|
406
463
|
const workflow = new glue.Workflow(stack, 'Workflow');
|
|
407
464
|
workflow.addOnDemandTrigger('OnDemandTrigger', {
|
|
408
|
-
actions: [
|
|
465
|
+
actions: [glue.Action.job(job)],
|
|
409
466
|
});
|
|
410
467
|
```
|
|
411
468
|
|
|
@@ -418,21 +475,35 @@ actions list using the job or crawler objects using conditional types.
|
|
|
418
475
|
|
|
419
476
|
#### **2. Scheduled Triggers**
|
|
420
477
|
|
|
421
|
-
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
425
|
-
without
|
|
426
|
-
|
|
427
|
-
|
|
478
|
+
Use `addScheduledTrigger` with a `TriggerSchedule` to fire on a cron schedule.
|
|
479
|
+
`TriggerSchedule.daily()` and `TriggerSchedule.weekly()` are convenience
|
|
480
|
+
factories; `TriggerSchedule.cron(...)` lets you build any schedule from the
|
|
481
|
+
[existing event Schedule class](https://docs.aws.amazon.com/cdk/api/v2/docs/aws-cdk-lib.aws_events.Schedule.html)
|
|
482
|
+
without writing raw cron expressions. The L2 extracts the expression that Glue
|
|
483
|
+
requires from the `TriggerSchedule`.
|
|
484
|
+
|
|
485
|
+
```ts
|
|
486
|
+
import * as cdk from 'aws-cdk-lib';
|
|
487
|
+
import * as iam from 'aws-cdk-lib/aws-iam';
|
|
488
|
+
declare const stack: cdk.Stack;
|
|
489
|
+
declare const role: iam.IRole;
|
|
490
|
+
declare const script: glue.Code;
|
|
491
|
+
const job = new glue.PySparkEtlJob(stack, 'Job', { role, script });
|
|
492
|
+
const workflow = new glue.Workflow(stack, 'Workflow');
|
|
493
|
+
|
|
494
|
+
workflow.addScheduledTrigger('WeeklyTrigger', {
|
|
495
|
+
actions: [glue.Action.job(job)],
|
|
496
|
+
schedule: glue.TriggerSchedule.weekly(),
|
|
497
|
+
});
|
|
498
|
+
```
|
|
428
499
|
|
|
429
|
-
#### **3.
|
|
500
|
+
#### **3. Event Triggers**
|
|
430
501
|
|
|
431
|
-
|
|
432
|
-
For batching triggers, you must specify `
|
|
433
|
-
triggers, `
|
|
434
|
-
defaults to 900 seconds, but you can override the window to align with
|
|
435
|
-
|
|
502
|
+
Use `addEventTrigger` for EventBridge event-based triggers. There are two types:
|
|
503
|
+
batching and non-batching. For batching triggers, you must specify `batchSize`.
|
|
504
|
+
For non-batching triggers, `batchSize` defaults to 1. For both, `batchWindow`
|
|
505
|
+
defaults to 900 seconds, but you can override the window to align with your
|
|
506
|
+
workload's requirements.
|
|
436
507
|
|
|
437
508
|
#### **4. Conditional Triggers**
|
|
438
509
|
|
|
@@ -451,13 +522,14 @@ certain types of data stores.
|
|
|
451
522
|
|
|
452
523
|
* **Networking - the CDK determines the best fit subnet for Glue connection
|
|
453
524
|
configuration**
|
|
454
|
-
|
|
455
|
-
|
|
456
|
-
`vpcSubnets`
|
|
525
|
+
Configure VPC placement through the `network` property, built with
|
|
526
|
+
`ConnectionNetwork.subnet(subnet)` to pin a specific subnet, or
|
|
527
|
+
`ConnectionNetwork.vpc(vpc, vpcSubnets?)` to let the L2 select one via the
|
|
528
|
+
existing
|
|
457
529
|
[EC2 Subnet Selection](https://docs.aws.amazon.com/cdk/api/v2/python/aws_cdk.aws_ec2/SubnetSelection.html)
|
|
458
|
-
library
|
|
459
|
-
|
|
460
|
-
|
|
530
|
+
library. A Glue connection targets a single subnet, so the first subnet of
|
|
531
|
+
the selection is used. The two factories are mutually exclusive, so a subnet
|
|
532
|
+
and a VPC can never be combined.
|
|
461
533
|
|
|
462
534
|
Pin the connection to a specific subnet:
|
|
463
535
|
|
|
@@ -469,7 +541,7 @@ new glue.Connection(this, 'MyConnection', {
|
|
|
469
541
|
// The security groups granting AWS Glue inbound access to the data source within the VPC
|
|
470
542
|
securityGroups: [securityGroup],
|
|
471
543
|
// The VPC subnet which contains the data source
|
|
472
|
-
subnet,
|
|
544
|
+
network: glue.ConnectionNetwork.subnet(subnet),
|
|
473
545
|
});
|
|
474
546
|
```
|
|
475
547
|
|
|
@@ -481,9 +553,8 @@ declare const vpc: ec2.Vpc;
|
|
|
481
553
|
new glue.Connection(this, 'MyConnection', {
|
|
482
554
|
type: glue.ConnectionType.NETWORK,
|
|
483
555
|
securityGroups: [securityGroup],
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
vpcSubnets: { subnetType: ec2.SubnetType.PRIVATE_WITH_EGRESS },
|
|
556
|
+
// vpcSubnets is optional - defaults to private subnets
|
|
557
|
+
network: glue.ConnectionNetwork.vpc(vpc, { subnetType: ec2.SubnetType.PRIVATE_WITH_EGRESS }),
|
|
487
558
|
});
|
|
488
559
|
```
|
|
489
560
|
|
|
@@ -496,7 +567,7 @@ declare const db: rds.DatabaseCluster;
|
|
|
496
567
|
new glue.Connection(this, "RdsConnection", {
|
|
497
568
|
type: glue.ConnectionType.JDBC,
|
|
498
569
|
securityGroups: [securityGroup],
|
|
499
|
-
subnet,
|
|
570
|
+
network: glue.ConnectionNetwork.subnet(subnet),
|
|
500
571
|
secret: db.secret,
|
|
501
572
|
properties: {
|
|
502
573
|
JDBC_CONNECTION_URL: `jdbc:mysql://${db.clusterEndpoint.socketAddress}/databasename`,
|
|
@@ -945,8 +1016,9 @@ new glue.S3Table(this, 'MyTable', {
|
|
|
945
1016
|
min: '2020-01-01',
|
|
946
1017
|
max: '2023-12-31',
|
|
947
1018
|
format: 'yyyy-MM-dd',
|
|
948
|
-
interval
|
|
949
|
-
|
|
1019
|
+
// `step` bundles interval + unit (supply both or neither). Optional at day
|
|
1020
|
+
// precision or coarser; required when the format is sub-day (e.g. hours).
|
|
1021
|
+
step: { interval: 1, intervalUnit: glue.DateIntervalUnit.DAYS },
|
|
950
1022
|
}),
|
|
951
1023
|
},
|
|
952
1024
|
});
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
# Glue Job Arguments
|
|
2
|
+
|
|
3
|
+
## Status
|
|
4
|
+
|
|
5
|
+
accepted
|
|
6
|
+
|
|
7
|
+
## Context
|
|
8
|
+
|
|
9
|
+
Every Glue job resource (`AWS::Glue::Job`) accepts a `DefaultArguments` map — a
|
|
10
|
+
flat `string → string` dictionary of `--flag`/value pairs that Glue passes to the
|
|
11
|
+
job script on every run. Some of these arguments are ordinary user configuration
|
|
12
|
+
(`--additional-python-modules`, `--enable-glue-datacatalog`, `--TempDir`, …), but
|
|
13
|
+
others are the wire form of features the CDK L2 models with strongly-typed props.
|
|
14
|
+
|
|
15
|
+
The L2 job constructs therefore populate `DefaultArguments` from two sources:
|
|
16
|
+
|
|
17
|
+
1. **Construct-managed arguments** — derived by the construct from typed props or
|
|
18
|
+
from the job class itself. Examples:
|
|
19
|
+
- `continuousLogging` → `--enable-continuous-cloudwatch-log`,
|
|
20
|
+
`--continuous-log-logGroup`, `--continuous-log-logStreamPrefix`,
|
|
21
|
+
`--continuous-log-conversionPattern`, `--enable-continuous-log-filter`
|
|
22
|
+
- `enableMetrics` → `--enable-metrics`
|
|
23
|
+
- `enableObservabilityMetrics` → `--enable-observability-metrics`
|
|
24
|
+
- `sparkUI` → `--enable-spark-ui`, `--spark-event-logs-path`
|
|
25
|
+
- `extraJars` / `extraJarsFirst` / `extraPythonFiles` / `extraFiles` →
|
|
26
|
+
`--extra-jars`, `--user-jars-first`, `--extra-py-files`, `--extra-files`
|
|
27
|
+
- `className` → `--class` (Scala only)
|
|
28
|
+
- the job language itself → `--job-language`
|
|
29
|
+
- `librarySet` → `library-set` (Python Shell only)
|
|
30
|
+
2. **`defaultArguments`** — the untyped escape-hatch map the user supplies directly,
|
|
31
|
+
for arguments the L2 does *not* model.
|
|
32
|
+
|
|
33
|
+
The two sources can collide. Before this decision, the collision was resolved
|
|
34
|
+
silently and inconsistently across job types:
|
|
35
|
+
|
|
36
|
+
- `SparkJob` / `PythonShellJob` merged as `{ ...managed, ...userDefaultArguments }`,
|
|
37
|
+
so the user value won — a user could pass
|
|
38
|
+
`defaultArguments: { '--enable-continuous-cloudwatch-log': 'false' }` and silently
|
|
39
|
+
turn off a secure default.
|
|
40
|
+
- `RayJob` merged the other way, so the construct value won — a user's
|
|
41
|
+
`defaultArguments` entry for a managed key was silently dropped.
|
|
42
|
+
|
|
43
|
+
Both behaviors are footguns: one weakens the construct's secure/observable defaults
|
|
44
|
+
without warning, the other ignores explicit user input without warning. Because Glue
|
|
45
|
+
enables continuous CloudWatch logging by default and that data can contain sensitive
|
|
46
|
+
runtime values (SQL, row data, stack traces), the "user silently wins" case is also a
|
|
47
|
+
security concern.
|
|
48
|
+
|
|
49
|
+
Separately, Glue itself reserves a handful of argument keys for its own internal use
|
|
50
|
+
(`--debug`, `--mode`, `--JOB_NAME`, `--endpoint`). These are never valid user input on
|
|
51
|
+
any job type.
|
|
52
|
+
|
|
53
|
+
## Constraints
|
|
54
|
+
|
|
55
|
+
- `DefaultArguments` is a single flat map on the L1; there is no separate channel to
|
|
56
|
+
distinguish "managed" from "user" keys once they are merged. Whatever the L2 does,
|
|
57
|
+
it must produce one merged map.
|
|
58
|
+
- Which keys are managed varies by job type: `--class` exists only for Scala jobs,
|
|
59
|
+
`library-set` only for Python Shell, `--enable-spark-ui` only for Spark, and so on.
|
|
60
|
+
A single global list would either over-reject (block a key that is a legitimate
|
|
61
|
+
escape hatch for a job type that doesn't manage it — e.g. `--extra-py-files` on a
|
|
62
|
+
job with no `extraPythonFiles` prop) or under-reject.
|
|
63
|
+
- Argument keys can be tokens (e.g. produced by `CfnJson`) that only resolve at
|
|
64
|
+
deploy time. String comparison cannot see through them at synthesis.
|
|
65
|
+
- The set of managed keys must not be tied to the set the construct *happens to emit*
|
|
66
|
+
for a given configuration: a prop that turns a feature off (`enableMetrics:
|
|
67
|
+
false`) emits nothing, but the key is still construct-managed and must stay reserved.
|
|
68
|
+
|
|
69
|
+
## Decision
|
|
70
|
+
|
|
71
|
+
**A managed argument has exactly one way to be configured: its typed prop.** Passing a
|
|
72
|
+
construct-managed or Glue-reserved key through `defaultArguments` throws a
|
|
73
|
+
`ValidationError` at synthesis time rather than silently winning or being dropped.
|
|
74
|
+
`defaultArguments` remains the escape hatch for every argument the L2 does not model.
|
|
75
|
+
|
|
76
|
+
### Data flow
|
|
77
|
+
|
|
78
|
+
There is a single sink for every construct-managed argument — the base-class method:
|
|
79
|
+
|
|
80
|
+
```ts
|
|
81
|
+
protected setManagedArgument(key: string, value?: string): void
|
|
82
|
+
```
|
|
83
|
+
|
|
84
|
+
It records `key` in the reserved set and, when `value !== undefined`, emits it. A
|
|
85
|
+
subclass calls it once per managed key, passing `undefined` when the feature is off or
|
|
86
|
+
unset — the key is reserved either way. Subclasses do not build local argument maps, so
|
|
87
|
+
this is the *only* way to emit a managed argument: declaration and emission happen in the
|
|
88
|
+
same call, and the reserved set therefore cannot drift from what is emitted.
|
|
89
|
+
|
|
90
|
+
Each job subclass, in its constructor:
|
|
91
|
+
|
|
92
|
+
1. Registers its managed arguments through `setManagedArgument` — directly, or through
|
|
93
|
+
the shared helpers `setupContinuousLogging` (all job types),
|
|
94
|
+
`nonExecutableCommonArguments` and `setupExtraCodeArguments` (Spark), and its own
|
|
95
|
+
`executableArguments` (`--job-language`; plus `--class` for Scala, `library-set` for
|
|
96
|
+
Python Shell).
|
|
97
|
+
2. Calls the base-class method:
|
|
98
|
+
|
|
99
|
+
```ts
|
|
100
|
+
protected mergeDefaultArguments(
|
|
101
|
+
defaultArguments?: { [key: string]: string },
|
|
102
|
+
): { [key: string]: string }
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
which validates the user-supplied `defaultArguments` against the accumulated reserved
|
|
106
|
+
set and returns the merged map, passed as `DefaultArguments` on the `CfnJob`.
|
|
107
|
+
|
|
108
|
+
`setManagedArgument` declares managed keys; `mergeDefaultArguments` validates and merges
|
|
109
|
+
them. Each is the sole choke point for its job.
|
|
110
|
+
|
|
111
|
+
Which keys a job type reserves falls out of which `setManagedArgument` calls its
|
|
112
|
+
constructor makes: only Scala jobs register `--class`, only Python Shell registers
|
|
113
|
+
`library-set`, only Spark registers `--enable-spark-ui`, and so on. Per-job-type scoping
|
|
114
|
+
is automatic — there is no separate list to maintain per type.
|
|
115
|
+
|
|
116
|
+
### The reserved set
|
|
117
|
+
|
|
118
|
+
The keys a user may not set through `defaultArguments` are the union of:
|
|
119
|
+
|
|
120
|
+
- **`GLUE_RESERVED_ARGUMENTS`** — `--debug`, `--mode`, `--JOB_NAME`, `--endpoint`. Owned
|
|
121
|
+
by the Glue service, reserved on every job type. This is the one static list, because
|
|
122
|
+
it is external to the constructs — nothing derives it from a prop.
|
|
123
|
+
- **`_managedArgumentKeys`** — every key that any `setManagedArgument` call registered on
|
|
124
|
+
this instance, whether or not a value was emitted for it.
|
|
125
|
+
|
|
126
|
+
### Validation rules (per user-supplied key)
|
|
127
|
+
|
|
128
|
+
For each key in `defaultArguments`:
|
|
129
|
+
|
|
130
|
+
1. If the key is an unresolved token, the conflict check is skipped (equality is
|
|
131
|
+
unknowable at synth time) and a warning
|
|
132
|
+
(`@aws-cdk/aws-glue-alpha:tokenJobArgumentKey`) is emitted. If it resolves to a
|
|
133
|
+
managed key at deploy time, the construct-managed value wins (see merge order).
|
|
134
|
+
2. If the key is in **`GLUE_RESERVED_ARGUMENTS`** → throw. Glue-reserved keys are
|
|
135
|
+
never emitted by the construct, so there is no construct value to reconcile against.
|
|
136
|
+
3. If the key is in the reserved set (`_managedArgumentKeys`):
|
|
137
|
+
- If the construct actually emitted a value for that key
|
|
138
|
+
(`Object.hasOwn(_managedArguments, key)`) and the supplied value is identical →
|
|
139
|
+
allowed. Passing the same value the construct would produce is not
|
|
140
|
+
contradictory; autocorrecting config is preferred over an error.
|
|
141
|
+
- Otherwise (different value, or the construct emitted nothing because the feature
|
|
142
|
+
is off) → throw.
|
|
143
|
+
4. Otherwise, the key is genuinely custom → allowed, flows through untouched.
|
|
144
|
+
|
|
145
|
+
`Object.hasOwn` is used deliberately instead of the `in` operator so that inherited
|
|
146
|
+
`Object.prototype` members (`toString`, `constructor`, `hasOwnProperty`, …) supplied as
|
|
147
|
+
argument keys are treated as ordinary custom keys rather than falsely matching a
|
|
148
|
+
managed key.
|
|
149
|
+
|
|
150
|
+
### Merge order
|
|
151
|
+
|
|
152
|
+
After validation, the result is:
|
|
153
|
+
|
|
154
|
+
```ts
|
|
155
|
+
return { ...defaultArguments, ...this._managedArguments };
|
|
156
|
+
```
|
|
157
|
+
|
|
158
|
+
Managed arguments are spread last, so they win on any residual overlap. By this point
|
|
159
|
+
the only overlaps that can remain are (a) exact-value matches allowed by rule 3, which
|
|
160
|
+
are indistinguishable either way, and (b) token keys from rule 1, for which
|
|
161
|
+
managed-wins is the documented and warned-about behavior.
|
|
162
|
+
|
|
163
|
+
### Related synthesis-time warnings
|
|
164
|
+
|
|
165
|
+
Two other warnings live in the same flow:
|
|
166
|
+
|
|
167
|
+
- **`@aws-cdk/aws-glue-alpha:unencryptedContinuousLogging`** — continuous logging is on
|
|
168
|
+
(explicitly or by default) but no `SecurityConfiguration` is attached, so driver /
|
|
169
|
+
executor logs land in an unencrypted, account-shared CloudWatch log group. We only
|
|
170
|
+
warn when *no* security configuration is attached at all, because
|
|
171
|
+
`ISecurityConfiguration` exposes only the name and we cannot introspect whether it
|
|
172
|
+
actually configures `cloudWatchEncryption` (avoiding false positives).
|
|
173
|
+
- **`@aws-cdk/aws-glue-alpha:plaintextJobArgumentSecret`** — a `defaultArguments` key
|
|
174
|
+
looks like a credential and holds a plaintext literal. `DefaultArguments` is emitted
|
|
175
|
+
verbatim into the template; secrets belong in AWS Secrets Manager.
|
|
176
|
+
|
|
177
|
+
## Alternatives
|
|
178
|
+
|
|
179
|
+
### Invert precedence so the construct always wins
|
|
180
|
+
|
|
181
|
+
Merge as `{ ...userDefaultArguments, ...managed }` everywhere (which is what `RayJob`
|
|
182
|
+
already did). This is more secure than "user wins" but still silent: a user who
|
|
183
|
+
deliberately sets a managed key via `defaultArguments` has it dropped with no
|
|
184
|
+
indication. It also still leaves two channels for one setting. Rejected in favor of a
|
|
185
|
+
single, explicit way to express each intent.
|
|
186
|
+
|
|
187
|
+
### Derive the reserved set from the emitted arguments
|
|
188
|
+
|
|
189
|
+
Compute conflicts from the keys the construct actually emits. Attractive: adding a typed
|
|
190
|
+
prop reserves its key automatically, with no separate list to maintain. Rejected: it ties
|
|
191
|
+
the reserved set to the current configuration. A prop that turns a feature off emits no
|
|
192
|
+
key, so `defaultArguments` could silently re-enable it (`enableMetrics: false` +
|
|
193
|
+
`defaultArguments: { '--enable-metrics': '' }`). The reserved set must include keys the
|
|
194
|
+
construct manages even when it emits nothing for them. That is why `setManagedArgument`
|
|
195
|
+
records the key regardless of value.
|
|
196
|
+
|
|
197
|
+
### A per-job-type list of managed keys
|
|
198
|
+
|
|
199
|
+
Give each job type an explicit `string[]` of the keys it manages, passed to the
|
|
200
|
+
validation step alongside the emitted map. Correct, but it names every managed key in
|
|
201
|
+
two places — the emission site (inside an `enabled ? {...} : {}` expression) and the
|
|
202
|
+
list — which drift: adding a typed prop requires updating the list too, and a forgotten
|
|
203
|
+
entry silently reopens the re-enable bypass with no compile-time signal. The
|
|
204
|
+
`setManagedArgument` accumulator collapses the two into one call, so the key is named
|
|
205
|
+
once.
|
|
206
|
+
|
|
207
|
+
### One global reserved list on the base class
|
|
208
|
+
|
|
209
|
+
A single list of every managed key across all job types. Rejected because it
|
|
210
|
+
over-rejects: it would block, for example, `--extra-py-files` on a job type that has no
|
|
211
|
+
`extraPythonFiles` prop, removing a legitimate escape hatch with no typed replacement.
|
|
212
|
+
Managed keys must be scoped per job type.
|
package/lib/catalog.js
CHANGED
|
@@ -74,7 +74,7 @@ var CatalogEncryptionMode;
|
|
|
74
74
|
* @see https://docs.aws.amazon.com/glue/latest/webapi/API_EncryptionAtRest.html
|
|
75
75
|
*/
|
|
76
76
|
class DataCatalogEncryptionAtRest {
|
|
77
|
-
static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.DataCatalogEncryptionAtRest", version: "2.
|
|
77
|
+
static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.DataCatalogEncryptionAtRest", version: "2.269.0-alpha.0" };
|
|
78
78
|
/**
|
|
79
79
|
* Disable encryption at rest for the Data Catalog.
|
|
80
80
|
*/
|
|
@@ -130,7 +130,7 @@ exports.DataCatalogEncryptionAtRest = DataCatalogEncryptionAtRest;
|
|
|
130
130
|
* construction, so a catalog either carries settings or it does not.
|
|
131
131
|
*/
|
|
132
132
|
class CatalogBase extends core_1.Resource {
|
|
133
|
-
static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.CatalogBase", version: "2.
|
|
133
|
+
static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.CatalogBase", version: "2.269.0-alpha.0" };
|
|
134
134
|
_encryptionKey;
|
|
135
135
|
_connectionPasswordKey;
|
|
136
136
|
get encryptionKey() {
|
|
@@ -224,7 +224,7 @@ let Catalog = (() => {
|
|
|
224
224
|
Catalog = _classThis = _classDescriptor.value;
|
|
225
225
|
if (_metadata) Object.defineProperty(_classThis, Symbol.metadata, { enumerable: true, configurable: true, writable: true, value: _metadata });
|
|
226
226
|
}
|
|
227
|
-
static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.Catalog", version: "2.
|
|
227
|
+
static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.Catalog", version: "2.269.0-alpha.0" };
|
|
228
228
|
/** Uniquely identifies this class. */
|
|
229
229
|
static PROPERTY_INJECTION_ID = '@aws-cdk.aws-glue-alpha.Catalog';
|
|
230
230
|
/**
|
package/lib/code.js
CHANGED
|
@@ -43,7 +43,7 @@ const helpers_internal_1 = require("aws-cdk-lib/core/lib/helpers-internal");
|
|
|
43
43
|
* Represents a Glue Job's Code assets (an asset can be a scripts, a jar, a python file or any other file).
|
|
44
44
|
*/
|
|
45
45
|
class Code {
|
|
46
|
-
static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.Code", version: "2.
|
|
46
|
+
static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.Code", version: "2.269.0-alpha.0" };
|
|
47
47
|
/**
|
|
48
48
|
* Job code as an S3 object.
|
|
49
49
|
* @param bucket The S3 bucket
|
|
@@ -68,7 +68,7 @@ exports.Code = Code;
|
|
|
68
68
|
class S3Code extends Code {
|
|
69
69
|
bucket;
|
|
70
70
|
key;
|
|
71
|
-
static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.S3Code", version: "2.
|
|
71
|
+
static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.S3Code", version: "2.269.0-alpha.0" };
|
|
72
72
|
constructor(bucket, key) {
|
|
73
73
|
super();
|
|
74
74
|
this.bucket = bucket;
|
|
@@ -91,7 +91,7 @@ exports.S3Code = S3Code;
|
|
|
91
91
|
class AssetCode extends Code {
|
|
92
92
|
path;
|
|
93
93
|
options;
|
|
94
|
-
static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.AssetCode", version: "2.
|
|
94
|
+
static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.AssetCode", version: "2.269.0-alpha.0" };
|
|
95
95
|
asset;
|
|
96
96
|
/**
|
|
97
97
|
* @param path The path to the Code file.
|
package/lib/connection.d.ts
CHANGED
|
@@ -198,6 +198,44 @@ export interface IConnection extends cdk.IResource {
|
|
|
198
198
|
*/
|
|
199
199
|
readonly connectionArn: string;
|
|
200
200
|
}
|
|
201
|
+
/**
|
|
202
|
+
* VPC network placement for a Glue `Connection`.
|
|
203
|
+
*
|
|
204
|
+
* A Glue connection targets a single subnet. Choose the placement with one of
|
|
205
|
+
* the mutually-exclusive factories — an explicit subnet, or a VPC to select one
|
|
206
|
+
* from — so a subnet paired with a VPC, or a subnet selection without a VPC,
|
|
207
|
+
* cannot be expressed.
|
|
208
|
+
*/
|
|
209
|
+
export declare class ConnectionNetwork {
|
|
210
|
+
/**
|
|
211
|
+
* Pin the connection to a specific subnet.
|
|
212
|
+
*
|
|
213
|
+
* @param subnet the subnet the connection targets.
|
|
214
|
+
*/
|
|
215
|
+
static subnet(subnet: ec2.ISubnet): ConnectionNetwork;
|
|
216
|
+
/**
|
|
217
|
+
* Select the connection's subnet from a VPC. Since a Glue connection targets
|
|
218
|
+
* a single subnet, the first subnet of the selection is used.
|
|
219
|
+
*
|
|
220
|
+
* @param vpc the VPC to select a subnet from.
|
|
221
|
+
* @param vpcSubnets which subnets to select from.
|
|
222
|
+
* @default vpcSubnets - private subnets
|
|
223
|
+
*/
|
|
224
|
+
static vpc(vpc: ec2.IVpc, vpcSubnets?: ec2.SubnetSelection): ConnectionNetwork;
|
|
225
|
+
/** @internal */
|
|
226
|
+
readonly _subnet?: ec2.ISubnet;
|
|
227
|
+
/** @internal */
|
|
228
|
+
readonly _vpc?: ec2.IVpc;
|
|
229
|
+
/** @internal */
|
|
230
|
+
readonly _vpcSubnets?: ec2.SubnetSelection;
|
|
231
|
+
private constructor();
|
|
232
|
+
/**
|
|
233
|
+
* Resolve the single subnet this network targets.
|
|
234
|
+
*
|
|
235
|
+
* @internal
|
|
236
|
+
*/
|
|
237
|
+
_resolveSubnet(scope: constructs.Construct): ec2.ISubnet | undefined;
|
|
238
|
+
}
|
|
201
239
|
/**
|
|
202
240
|
* Base Connection Options
|
|
203
241
|
*/
|
|
@@ -243,33 +281,15 @@ export interface ConnectionOptions {
|
|
|
243
281
|
*/
|
|
244
282
|
readonly securityGroups?: ec2.ISecurityGroup[];
|
|
245
283
|
/**
|
|
246
|
-
* The VPC
|
|
247
|
-
*
|
|
248
|
-
* Mutually exclusive with `vpc`: provide `subnet` to pin the connection to a
|
|
249
|
-
* specific subnet, or provide `vpc` (optionally with `vpcSubnets`) to let the
|
|
250
|
-
* CDK select one for you.
|
|
251
|
-
*
|
|
252
|
-
* @default - no subnet, unless `vpc` is provided
|
|
253
|
-
*/
|
|
254
|
-
readonly subnet?: ec2.ISubnet;
|
|
255
|
-
/**
|
|
256
|
-
* The VPC to connect to resources within. When provided, the CDK selects a
|
|
257
|
-
* subnet from this VPC using `vpcSubnets`. A Glue connection targets a single
|
|
258
|
-
* subnet, so the first subnet of the selection is used.
|
|
284
|
+
* The VPC network placement for this connection, so it can reach resources
|
|
285
|
+
* inside a VPC. See more at https://docs.aws.amazon.com/glue/latest/dg/start-connecting.html.
|
|
259
286
|
*
|
|
260
|
-
*
|
|
287
|
+
* Build it with `ConnectionNetwork.subnet(subnet)` to pin a specific subnet,
|
|
288
|
+
* or `ConnectionNetwork.vpc(vpc, vpcSubnets?)` to let the CDK select one.
|
|
261
289
|
*
|
|
262
|
-
* @default - no VPC
|
|
290
|
+
* @default - no VPC network placement
|
|
263
291
|
*/
|
|
264
|
-
readonly
|
|
265
|
-
/**
|
|
266
|
-
* Which subnets of `vpc` to select the connection subnet from. Only used when
|
|
267
|
-
* `vpc` is provided. Since a Glue connection targets a single subnet, the
|
|
268
|
-
* first subnet of the selection is used.
|
|
269
|
-
*
|
|
270
|
-
* @default - private subnets
|
|
271
|
-
*/
|
|
272
|
-
readonly vpcSubnets?: ec2.SubnetSelection;
|
|
292
|
+
readonly network?: ConnectionNetwork;
|
|
273
293
|
}
|
|
274
294
|
/**
|
|
275
295
|
* Construction properties for `Connection`
|
|
@@ -306,11 +326,6 @@ export declare class Connection extends cdk.Resource implements IConnection {
|
|
|
306
326
|
private readonly properties;
|
|
307
327
|
private readonly resource;
|
|
308
328
|
constructor(scope: constructs.Construct, id: string, props: ConnectionProps);
|
|
309
|
-
/**
|
|
310
|
-
* Determines the single subnet the connection should target, either from an
|
|
311
|
-
* explicit `subnet` or by selecting one from `vpc`.
|
|
312
|
-
*/
|
|
313
|
-
private resolveSubnet;
|
|
314
329
|
/**
|
|
315
330
|
* The name of the connection
|
|
316
331
|
*/
|