@aws-cdk/aws-glue-alpha 2.265.0-alpha.0 → 2.267.0-alpha.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/.jsii +982 -776
  2. package/.jsii.tabl.json.gz +0 -0
  3. package/.warnings.jsii.js +25 -91
  4. package/README.md +158 -72
  5. package/awslint.json +6 -1
  6. package/lib/catalog.js +3 -3
  7. package/lib/code.js +3 -3
  8. package/lib/connection.d.ts +41 -1
  9. package/lib/connection.js +48 -9
  10. package/lib/data-format.d.ts +2 -2
  11. package/lib/data-format.js +8 -8
  12. package/lib/data-quality-ruleset.d.ts +28 -3
  13. package/lib/data-quality-ruleset.js +37 -5
  14. package/lib/database.d.ts +1 -1
  15. package/lib/database.js +2 -2
  16. package/lib/external-table.js +6 -4
  17. package/lib/index.d.ts +0 -1
  18. package/lib/index.js +1 -2
  19. package/lib/jobs/job.d.ts +1 -24
  20. package/lib/jobs/job.js +3 -3
  21. package/lib/jobs/pyspark-etl-job.d.ts +2 -1
  22. package/lib/jobs/pyspark-etl-job.js +10 -5
  23. package/lib/jobs/pyspark-flex-etl-job.d.ts +3 -2
  24. package/lib/jobs/pyspark-flex-etl-job.js +7 -6
  25. package/lib/jobs/pyspark-streaming-job.d.ts +2 -1
  26. package/lib/jobs/pyspark-streaming-job.js +6 -5
  27. package/lib/jobs/python-shell-job.js +1 -1
  28. package/lib/jobs/ray-job.d.ts +8 -0
  29. package/lib/jobs/ray-job.js +3 -7
  30. package/lib/jobs/scala-spark-etl-job.js +4 -8
  31. package/lib/jobs/scala-spark-flex-etl-job.js +4 -4
  32. package/lib/jobs/scala-spark-streaming-job.js +4 -8
  33. package/lib/jobs/spark-job.d.ts +26 -0
  34. package/lib/jobs/spark-job.js +2 -2
  35. package/lib/partition-projection.d.ts +8 -8
  36. package/lib/partition-projection.js +49 -2
  37. package/lib/s3-table.d.ts +83 -39
  38. package/lib/s3-table.js +136 -86
  39. package/lib/schema.d.ts +26 -6
  40. package/lib/schema.js +74 -82
  41. package/lib/security-configuration.d.ts +41 -47
  42. package/lib/security-configuration.js +83 -25
  43. package/lib/storage-parameter.d.ts +14 -3
  44. package/lib/storage-parameter.js +16 -6
  45. package/lib/table-base.d.ts +34 -1
  46. package/lib/table-base.js +33 -2
  47. package/lib/triggers/trigger-options.js +1 -1
  48. package/lib/triggers/workflow.js +2 -2
  49. package/package.json +7 -7
  50. package/lib/table-deprecated.d.ts +0 -13
  51. package/lib/table-deprecated.js +0 -70
package/README.md CHANGED
@@ -75,7 +75,9 @@ similar constructors. ETL jobs default to the G2 worker type, but you can
75
75
  override this default with other supported worker type values (G1, G2, G4
76
76
  and G8). ETL jobs defaults to Glue version 4.0, which you can override to 3.0.
77
77
  The following ETL features are enabled by default:
78
- `—enable-metrics, —enable-spark-ui, —enable-continuous-cloudwatch-log.`
78
+ `—enable-metrics, —enable-continuous-cloudwatch-log.`
79
+ The Spark UI (`—enable-spark-ui`) is off by default; enable it by setting the
80
+ `sparkUI` prop.
79
81
  You can find more details about version, worker type and other features in
80
82
  [Glue's public documentation](https://docs.aws.amazon.com/glue/latest/dg/aws-glue-api-jobs-job.html).
81
83
 
@@ -115,7 +117,10 @@ new glue.PySparkEtlJob(stack, 'PySparkETLJob', {
115
117
  script,
116
118
  glueVersion: glue.GlueVersion.V5_1,
117
119
  continuousLogging: { enabled: false },
118
- workerType: glue.WorkerType.G_2X,
120
+ workerConfiguration: {
121
+ workerType: glue.WorkerType.G_2X,
122
+ numberOfWorkers: 2,
123
+ },
119
124
  maxConcurrentRuns: 100,
120
125
  timeout: cdk.Duration.hours(2),
121
126
  connections: [glue.Connection.fromConnectionName(stack, 'Connection', 'connectionName')],
@@ -125,7 +130,6 @@ new glue.PySparkEtlJob(stack, 'PySparkETLJob', {
125
130
  SecondTagName: 'SecondTagValue',
126
131
  XTagName: 'XTagValue',
127
132
  },
128
- numberOfWorkers: 2,
129
133
  maxRetries: 2,
130
134
  });
131
135
  ```
@@ -135,11 +139,12 @@ new glue.PySparkEtlJob(stack, 'PySparkETLJob', {
135
139
  Streaming jobs are similar to ETL jobs, except that they perform ETL on data
136
140
  streams using the Apache Spark Structured Streaming framework. Some Spark
137
141
  job features are not available to Streaming ETL jobs. They support Scala
138
- and pySpark languages. PySpark streaming jobs default Python 3.9,
139
- which you can override with any non-deprecated version of Python. It
142
+ and pySpark languages. PySpark streaming jobs run on Python 3. It
140
143
  defaults to the G2 worker type and Glue 4.0, both of which you can override.
141
144
  The following best practice features are enabled by default:
142
- `—enable-metrics, —enable-spark-ui, —enable-continuous-cloudwatch-log`.
145
+ `—enable-metrics, —enable-continuous-cloudwatch-log`.
146
+ The Spark UI (`—enable-spark-ui`) is off by default; enable it by setting the
147
+ `sparkUI` prop.
143
148
 
144
149
  Reference the pyspark-streaming-jobs.test.ts and scalaspark-streaming-jobs.test.ts
145
150
  unit tests for examples of required-only and optional job parameters when creating
@@ -171,7 +176,10 @@ new glue.PySparkStreamingJob(stack, 'PySparkStreamingJob', {
171
176
  script,
172
177
  glueVersion: glue.GlueVersion.V5_1,
173
178
  continuousLogging: { enabled: false },
174
- workerType: glue.WorkerType.G_2X,
179
+ workerConfiguration: {
180
+ workerType: glue.WorkerType.G_2X,
181
+ numberOfWorkers: 2,
182
+ },
175
183
  maxConcurrentRuns: 100,
176
184
  timeout: cdk.Duration.hours(2),
177
185
  connections: [glue.Connection.fromConnectionName(stack, 'Connection', 'connectionName')],
@@ -181,7 +189,6 @@ new glue.PySparkStreamingJob(stack, 'PySparkStreamingJob', {
181
189
  SecondTagName: 'SecondTagValue',
182
190
  XTagName: 'XTagValue',
183
191
  },
184
- numberOfWorkers: 2,
185
192
  maxRetries: 2,
186
193
  });
187
194
  ```
@@ -192,7 +199,9 @@ The flexible execution class is appropriate for non-urgent jobs such as
192
199
  pre-production jobs, testing, and one-time data loads. Flexible jobs default
193
200
  to Glue version 5.0 and worker type `G_2X`. The following best practice
194
201
  features are enabled by default:
195
- `—enable-metrics, —enable-spark-ui, —enable-continuous-cloudwatch-log`
202
+ `—enable-metrics, —enable-continuous-cloudwatch-log`
203
+ The Spark UI (`—enable-spark-ui`) is off by default; enable it by setting the
204
+ `sparkUI` prop.
196
205
 
197
206
  Reference the pyspark-flex-etl-jobs.test.ts and scalaspark-flex-etl-jobs.test.ts
198
207
  unit tests for examples of required-only and optional job parameters when creating
@@ -217,14 +226,17 @@ import * as iam from 'aws-cdk-lib/aws-iam';
217
226
  declare const stack: cdk.Stack;
218
227
  declare const role: iam.IRole;
219
228
  declare const script: glue.Code;
220
- new glue.PySparkEtlJob(stack, 'pySparkEtlJob', {
221
- jobName: 'pySparkEtlJob',
229
+ new glue.PySparkFlexEtlJob(stack, 'pySparkFlexEtlJob', {
230
+ jobName: 'pySparkFlexEtlJob',
222
231
  description: 'This is a description',
223
232
  role,
224
233
  script,
225
234
  glueVersion: glue.GlueVersion.V5_1,
226
235
  continuousLogging: { enabled: false },
227
- workerType: glue.WorkerType.G_2X,
236
+ workerConfiguration: {
237
+ workerType: glue.WorkerType.G_2X,
238
+ numberOfWorkers: 2,
239
+ },
228
240
  maxConcurrentRuns: 100,
229
241
  timeout: cdk.Duration.hours(2),
230
242
  connections: [glue.Connection.fromConnectionName(stack, 'Connection', 'connectionName')],
@@ -234,7 +246,6 @@ new glue.PySparkEtlJob(stack, 'pySparkEtlJob', {
234
246
  SecondTagName: 'SecondTagValue',
235
247
  XTagName: 'XTagValue',
236
248
  },
237
- numberOfWorkers: 2,
238
249
  maxRetries: 2,
239
250
  });
240
251
  ```
@@ -274,14 +285,13 @@ declare const extraPythonFile: glue.Code;
274
285
  new glue.PythonShellJob(stack, 'PythonShellJob', {
275
286
  jobName: 'PythonShellJobCustomName',
276
287
  description: 'This is a description',
277
- pythonVersion: glue.PythonVersion.TWO,
288
+ pythonVersion: glue.PythonVersion.THREE_NINE,
278
289
  maxCapacity: glue.MaxCapacity.DPU_1,
279
290
  role,
280
291
  script,
281
292
  extraPythonFiles: [extraPythonFile],
282
- glueVersion: glue.GlueVersion.V2_0,
293
+ glueVersion: glue.GlueVersion.V3_0,
283
294
  continuousLogging: { enabled: false },
284
- workerType: glue.WorkerType.G_2X,
285
295
  maxConcurrentRuns: 100,
286
296
  timeout: cdk.Duration.hours(2),
287
297
  connections: [glue.Connection.fromConnectionName(stack, 'Connection', 'connectionName')],
@@ -291,7 +301,6 @@ new glue.PythonShellJob(stack, 'PythonShellJob', {
291
301
  SecondTagName: 'SecondTagValue',
292
302
  XTagName: 'XTagValue',
293
303
  },
294
- numberOfWorkers: 2,
295
304
  maxRetries: 2,
296
305
  });
297
306
  ```
@@ -355,10 +364,12 @@ new glue.PySparkEtlJob(stack, 'PySparkETLJob', {
355
364
 
356
365
  ### Uploading scripts from the CDK app repository to S3
357
366
 
358
- Similar to other L2 constructs, the Glue L2 automates uploading / updating
359
- scripts to S3 via an optional fromAsset parameter pointing to a script
360
- in the local file structure. You provide the existing S3 bucket and
361
- path to which you'd like the script to be uploaded.
367
+ Similar to other L2 constructs, the Glue L2 automates uploading local
368
+ scripts to S3. Use `glue.Code.fromAsset(path)` to point at a script in your
369
+ local file structure; it is uploaded to the CDK-managed asset bucket. To
370
+ reference a script that already exists in S3, use
371
+ `glue.Code.fromBucket(bucket, key)`, which performs no upload. A `script` is
372
+ required for every job.
362
373
 
363
374
  Reference the unit tests for examples of repo and S3 code target examples.
364
375
 
@@ -370,15 +381,34 @@ jobs, and triggers. Standalone triggers are an anti-pattern, so you must
370
381
  create triggers from within a workflow using the L2 construct.
371
382
 
372
383
  Within a workflow object, there are functions to create different
373
- types of triggers with actions and predicates. You then add those triggers
374
- to jobs.
384
+ types of triggers with actions and predicates. You add triggers to the
385
+ workflow, and each trigger references the jobs or crawlers it runs as its
386
+ actions.
375
387
 
376
- StartOnCreation defaults to true for all trigger types, but you can
377
- override it if you prefer for your trigger not to start on creation.
388
+ `startOnCreation` applies to scheduled triggers (and, via
389
+ `ConditionalTriggerOptions`, conditional triggers) only. It defaults to `false`,
390
+ but you can override it if you prefer for your trigger to start on creation.
378
391
 
379
392
  Reference the workflow-triggers.test.ts unit tests for examples of creating
380
393
  workflows and triggers.
381
394
 
395
+ ```ts
396
+ import * as cdk from 'aws-cdk-lib';
397
+ import * as iam from 'aws-cdk-lib/aws-iam';
398
+ declare const stack: cdk.Stack;
399
+ declare const role: iam.IRole;
400
+ declare const script: glue.Code;
401
+
402
+ // Create a job to run from the workflow
403
+ const job = new glue.PySparkEtlJob(stack, 'Job', { role, script });
404
+
405
+ // Create a workflow and add a trigger that runs the job
406
+ const workflow = new glue.Workflow(stack, 'Workflow');
407
+ workflow.addOnDemandTrigger('OnDemandTrigger', {
408
+ actions: [{ job }],
409
+ });
410
+ ```
411
+
382
412
  #### **1. On-Demand Triggers**
383
413
 
384
414
  On-demand triggers can start glue jobs or crawlers. This construct provides
@@ -389,7 +419,7 @@ actions list using the job or crawler objects using conditional types.
389
419
  #### **2. Scheduled Triggers**
390
420
 
391
421
  You can create scheduled triggers using cron expressions. This construct
392
- provides daily, weekly, and monthly convenience functions,
422
+ provides daily and weekly convenience functions,
393
423
  as well as a custom function that allows you to create your own
394
424
  custom timing using the [existing event Schedule class](https://docs.aws.amazon.com/cdk/api/v2/docs/aws-cdk-lib.aws_events.Schedule.html)
395
425
  without having to build your own cron expressions. The L2 extracts
@@ -415,18 +445,21 @@ A `Connection` allows Glue jobs, crawlers and development endpoints to access
415
445
  certain types of data stores.
416
446
 
417
447
  * **Secrets Management**
418
- You must specify JDBC connection credentials in Secrets Manager and
419
- provide the Secrets Manager Key name as a property to the job connection.
448
+ Manage JDBC connection credentials in Secrets Manager and pass the secret
449
+ to the connection via the `secret` property (see the example below), rather
450
+ than embedding credentials in `properties`.
420
451
 
421
452
  * **Networking - the CDK determines the best fit subnet for Glue connection
422
453
  configuration**
423
- The prior version of the glue-alpha-module requires the developer to
424
- specify the subnet of the Connection when it’s defined. Now, you can still
425
- specify the specific subnet you want to use, but are no longer required
426
- to. You are only required to provide a VPC and either a public or private
427
- subnet selection. Without a specific subnet provided, the L2 leverages the
428
- existing [EC2 Subnet Selection](https://docs.aws.amazon.com/cdk/api/v2/python/aws_cdk.aws_ec2/SubnetSelection.html)
429
- library to make the best choice selection for the subnet.
454
+ You can specify the exact subnet of the Connection when it's defined, but
455
+ you are not required to. Instead, you can provide a `vpc` and, optionally, a
456
+ `vpcSubnets` selection, and the L2 leverages the existing
457
+ [EC2 Subnet Selection](https://docs.aws.amazon.com/cdk/api/v2/python/aws_cdk.aws_ec2/SubnetSelection.html)
458
+ library to make the best choice selection for the subnet. A Glue connection
459
+ targets a single subnet, so the first subnet of the selection is used.
460
+ `subnet` and `vpc` are mutually exclusive.
461
+
462
+ Pin the connection to a specific subnet:
430
463
 
431
464
  ```ts
432
465
  declare const securityGroup: ec2.SecurityGroup;
@@ -440,7 +473,21 @@ new glue.Connection(this, 'MyConnection', {
440
473
  });
441
474
  ```
442
475
 
443
- For RDS `Connection` by JDBC, it is recommended to manage credentials using AWS Secrets Manager. To use Secret, specify `SECRET_ID` in `properties` like the following code. Note that in this case, the subnet must have a route to the AWS Secrets Manager VPC endpoint or to the AWS Secrets Manager endpoint through a NAT gateway.
476
+ Or let the CDK select a subnet from a VPC:
477
+
478
+ ```ts
479
+ declare const securityGroup: ec2.SecurityGroup;
480
+ declare const vpc: ec2.Vpc;
481
+ new glue.Connection(this, 'MyConnection', {
482
+ type: glue.ConnectionType.NETWORK,
483
+ securityGroups: [securityGroup],
484
+ vpc,
485
+ // Optional - defaults to private subnets
486
+ vpcSubnets: { subnetType: ec2.SubnetType.PRIVATE_WITH_EGRESS },
487
+ });
488
+ ```
489
+
490
+ For RDS `Connection` by JDBC, it is recommended to manage credentials using AWS Secrets Manager. Pass the secret via the `secret` property: Glue reads the credentials at runtime through the connection's `SECRET_ID`, so the secret value never enters the template. Note that in this case, the subnet must have a route to the AWS Secrets Manager VPC endpoint or to the AWS Secrets Manager endpoint through a NAT gateway.
444
491
 
445
492
  ```ts
446
493
  declare const securityGroup: ec2.SecurityGroup;
@@ -450,18 +497,18 @@ new glue.Connection(this, "RdsConnection", {
450
497
  type: glue.ConnectionType.JDBC,
451
498
  securityGroups: [securityGroup],
452
499
  subnet,
500
+ secret: db.secret,
453
501
  properties: {
454
502
  JDBC_CONNECTION_URL: `jdbc:mysql://${db.clusterEndpoint.socketAddress}/databasename`,
455
503
  JDBC_ENFORCE_SSL: "false",
456
- SECRET_ID: db.secret!.secretName,
457
504
  },
458
505
  });
459
506
  ```
460
507
 
461
- Connection `properties` are emitted verbatim into the CloudFormation template, so
462
- any credential placed there in plaintext is stored in plaintext in the template,
463
- `cdk.out`, and source control. Reference a Secrets Manager secret through
464
- `SECRET_ID` (as above) instead. If a property key looks like a credential (for
508
+ Prefer the `secret` property over placing credentials in `properties`. Connection
509
+ `properties` are emitted verbatim into the CloudFormation template, so any
510
+ credential placed there in plaintext is stored in plaintext in the template,
511
+ `cdk.out`, and source control. If a property key looks like a credential (for
465
512
  example `PASSWORD`, `SECRET`, or `TOKEN`) and holds a plaintext literal, the
466
513
  construct emits a synthesis-time warning.
467
514
 
@@ -473,32 +520,29 @@ See [Adding a Connection to Your Data Store](https://docs.aws.amazon.com/glue/la
473
520
 
474
521
  A `SecurityConfiguration` is a set of security properties that can be used by AWS Glue to encrypt data at rest.
475
522
 
523
+ Each encryption config is built with a factory that pairs the encryption mode
524
+ with its key, so illegal combinations (such as an S3-managed encryption carrying
525
+ a KMS key) cannot be expressed:
526
+
476
527
  ```ts
477
528
  new glue.SecurityConfiguration(this, 'MySecurityConfiguration', {
478
- cloudWatchEncryption: {
479
- mode: glue.CloudWatchEncryptionMode.KMS,
480
- },
481
- jobBookmarksEncryption: {
482
- mode: glue.JobBookmarksEncryptionMode.CLIENT_SIDE_KMS,
483
- },
484
- s3Encryption: {
485
- mode: glue.S3EncryptionMode.KMS,
486
- },
529
+ cloudWatchEncryption: glue.CloudWatchEncryption.kms(),
530
+ jobBookmarksEncryption: glue.JobBookmarksEncryption.clientSideKms(),
531
+ s3Encryption: glue.S3Encryption.kms(),
487
532
  });
488
533
  ```
489
534
 
490
- By default, a shared KMS key is created for use with the encryption configurations that require one. You can also supply your own key for each encryption config, for example, for CloudWatch encryption:
535
+ By default, a shared KMS key is created for use with the encryption configurations that require one. You can also supply your own key to any factory, for example, for CloudWatch encryption:
491
536
 
492
537
  ```ts
493
538
  declare const key: kms.Key;
494
539
  new glue.SecurityConfiguration(this, 'MySecurityConfiguration', {
495
- cloudWatchEncryption: {
496
- mode: glue.CloudWatchEncryptionMode.KMS,
497
- kmsKey: key,
498
- },
540
+ cloudWatchEncryption: glue.CloudWatchEncryption.kms(key),
499
541
  });
500
542
  ```
501
543
 
544
+ Use `glue.S3Encryption.s3Managed()` for S3-managed (SSE-S3) encryption, which takes no key.
545
+
502
546
  See [documentation](https://docs.aws.amazon.com/glue/latest/dg/encryption-security-configuration.html) for more info for Glue encrypting data written by Crawlers, Jobs, and Development Endpoints.
503
547
 
504
548
  ## Catalog
@@ -560,7 +604,7 @@ new glue.Catalog(this, 'MyCatalog', {
560
604
  ### Encryption at rest
561
605
 
562
606
  Configure Data Catalog encryption at rest through the `encryptionAtRest` option
563
- (on `Catalog.encryptAccount`, the `Catalog` constructor, or the import factories).
607
+ (on `Catalog.encryptAccount` or the `Catalog` constructor).
564
608
  It accepts a `DataCatalogEncryptionAtRest` describing the mode:
565
609
 
566
610
  ```ts
@@ -697,13 +741,13 @@ new glue.S3Table(this, 'MyTable', {
697
741
  });
698
742
  ```
699
743
 
700
- By default, a S3 bucket will be created to store the table's data but you can manually pass the `bucket` and `s3Prefix`:
744
+ By default, a S3 bucket will be created to store the table's data but you can bring your own with `S3TableStorage.fromBucket` and set an `s3Prefix`:
701
745
 
702
746
  ```ts
703
747
  declare const myBucket: s3.Bucket;
704
748
  declare const myDatabase: glue.Database;
705
749
  new glue.S3Table(this, 'MyTable', {
706
- bucket: myBucket,
750
+ storage: glue.S3TableStorage.fromBucket(myBucket),
707
751
  s3Prefix: 'my-table/',
708
752
  // ...
709
753
  database: myDatabase,
@@ -814,7 +858,7 @@ new glue.S3Table(this, 'MyTable', {
814
858
  Alternatively, you can call the `addPartitionIndex()` function on a table:
815
859
 
816
860
  ```ts
817
- declare const myTable: glue.Table;
861
+ declare const myTable: glue.S3Table;
818
862
  myTable.addPartitionIndex({
819
863
  indexName: 'my-index',
820
864
  keyNames: ['year'],
@@ -1047,18 +1091,38 @@ new glue.ExternalTable(this, 'MyTable', {
1047
1091
  });
1048
1092
  ```
1049
1093
 
1094
+ ## Data Quality Ruleset
1095
+
1096
+ A `DataQualityRuleset` defines a set of data quality rules — authored in Glue's
1097
+ Data Quality Definition Language (DQDL) — that are evaluated against a table in
1098
+ the Data Catalog.
1099
+
1100
+ ```ts
1101
+ new glue.DataQualityRuleset(this, 'MyRuleset', {
1102
+ rulesetName: 'my_ruleset',
1103
+ dqdl: glue.Dqdl.fromString('Rules = [ RowCount > 100, IsComplete "order_id" ]'),
1104
+ targetTable: new glue.DataQualityTargetTable('my_database', 'my_table'),
1105
+ });
1106
+ ```
1107
+
1108
+ Build the DQDL document with `Dqdl.fromString(...)`. Glue parses and validates the
1109
+ DQDL when the ruleset is deployed; see the
1110
+ [DQDL reference](https://docs.aws.amazon.com/glue/latest/dg/dqdl.html) for the
1111
+ full rule syntax.
1112
+
1050
1113
  ## [Encryption](https://docs.aws.amazon.com/athena/latest/ug/encryption.html)
1051
1114
 
1052
- When the table creates its own S3 bucket (i.e. you do not pass an explicit `bucket`), that bucket enforces SSL: a bucket policy denies any request made over plain HTTP. If you provide your own bucket, enabling `enforceSSL` on it is your responsibility.
1115
+ When the table creates its own S3 bucket (`S3TableStorage.managedBucket`, the default), that bucket enforces SSL: a bucket policy denies any request made over plain HTTP. If you bring your own bucket with `S3TableStorage.fromBucket`, enabling `enforceSSL` on it is your responsibility.
1053
1116
 
1054
- You can enable encryption on a Table's data:
1117
+ Server-side encryption applies only to a bucket the table manages. Choose it with
1118
+ `storage: glue.S3TableStorage.managedBucket(...)`:
1055
1119
 
1056
1120
  * [S3Managed](https://docs.aws.amazon.com/AmazonS3/latest/dev/UsingServerSideEncryption.html) - (default) Server side encryption (`SSE-S3`) with an Amazon S3-managed key.
1057
1121
 
1058
1122
  ```ts
1059
1123
  declare const myDatabase: glue.Database;
1060
1124
  new glue.S3Table(this, 'MyTable', {
1061
- encryption: glue.TableEncryption.S3_MANAGED,
1125
+ storage: glue.S3TableStorage.managedBucket(glue.S3TableEncryption.s3Managed()),
1062
1126
  // ...
1063
1127
  database: myDatabase,
1064
1128
  columns: [{
@@ -1075,7 +1139,7 @@ new glue.S3Table(this, 'MyTable', {
1075
1139
  declare const myDatabase: glue.Database;
1076
1140
  // KMS key is created automatically
1077
1141
  new glue.S3Table(this, 'MyTable', {
1078
- encryption: glue.TableEncryption.KMS,
1142
+ storage: glue.S3TableStorage.managedBucket(glue.S3TableEncryption.kms()),
1079
1143
  // ...
1080
1144
  database: myDatabase,
1081
1145
  columns: [{
@@ -1087,8 +1151,7 @@ new glue.S3Table(this, 'MyTable', {
1087
1151
 
1088
1152
  // with an explicit KMS key
1089
1153
  new glue.S3Table(this, 'MyTable', {
1090
- encryption: glue.TableEncryption.KMS,
1091
- encryptionKey: new kms.Key(this, 'MyKey'),
1154
+ storage: glue.S3TableStorage.managedBucket(glue.S3TableEncryption.kms(new kms.Key(this, 'MyKey'))),
1092
1155
  // ...
1093
1156
  database: myDatabase,
1094
1157
  columns: [{
@@ -1104,7 +1167,7 @@ new glue.S3Table(this, 'MyTable', {
1104
1167
  ```ts
1105
1168
  declare const myDatabase: glue.Database;
1106
1169
  new glue.S3Table(this, 'MyTable', {
1107
- encryption: glue.TableEncryption.KMS_MANAGED,
1170
+ storage: glue.S3TableStorage.managedBucket(glue.S3TableEncryption.kmsManaged()),
1108
1171
  // ...
1109
1172
  database: myDatabase,
1110
1173
  columns: [{
@@ -1115,13 +1178,13 @@ new glue.S3Table(this, 'MyTable', {
1115
1178
  });
1116
1179
  ```
1117
1180
 
1118
- * [ClientSideKms](https://docs.aws.amazon.com/AmazonS3/latest/dev/UsingClientSideEncryption.html#client-side-encryption-kms-managed-master-key-intro) - Client-side encryption (`CSE-KMS`) with an AWS KMS Key managed by the account owner.
1181
+ Client-side encryption ([CSE-KMS](https://docs.aws.amazon.com/AmazonS3/latest/dev/UsingClientSideEncryption.html#client-side-encryption-kms-managed-master-key-intro)) is independent of the bucket's server-side encryption and works with either a managed or an existing bucket. Configure it with `clientSideEncryption`:
1119
1182
 
1120
1183
  ```ts
1121
1184
  declare const myDatabase: glue.Database;
1122
1185
  // KMS key is created automatically
1123
1186
  new glue.S3Table(this, 'MyTable', {
1124
- encryption: glue.TableEncryption.CLIENT_SIDE_KMS,
1187
+ clientSideEncryption: glue.TableClientSideEncryption.kms(),
1125
1188
  // ...
1126
1189
  database: myDatabase,
1127
1190
  columns: [{
@@ -1133,8 +1196,7 @@ new glue.S3Table(this, 'MyTable', {
1133
1196
 
1134
1197
  // with an explicit KMS key
1135
1198
  new glue.S3Table(this, 'MyTable', {
1136
- encryption: glue.TableEncryption.CLIENT_SIDE_KMS,
1137
- encryptionKey: new kms.Key(this, 'MyKey'),
1199
+ clientSideEncryption: glue.TableClientSideEncryption.kms(new kms.Key(this, 'MyKey')),
1138
1200
  // ...
1139
1201
  database: myDatabase,
1140
1202
  columns: [{
@@ -1145,7 +1207,29 @@ new glue.S3Table(this, 'MyTable', {
1145
1207
  });
1146
1208
  ```
1147
1209
 
1148
- *Note: you cannot provide a `Bucket` when creating the `S3Table` if you wish to use server-side encryption (`KMS`, `KMS_MANAGED` or `S3_MANAGED`)*.
1210
+ To store the table's data in an existing bucket, use `glue.S3TableStorage.fromBucket(bucket)`. CDK does not manage that bucket's server-side encryption, so an encryption choice can never be paired with a provided bucket — but client-side encryption still applies.
1211
+
1212
+ ### Marking table data as encrypted
1213
+
1214
+ Both `S3Table` and `ExternalTable` set the `has_encrypted_data` table parameter, which
1215
+ Athena reads when querying client-side (`CSE-KMS`) encrypted datasets. It defaults to `true`.
1216
+ Set `hasEncryptedData` to `false` when the underlying data is not encrypted:
1217
+
1218
+ ```ts
1219
+ declare const myDatabase: glue.Database;
1220
+ new glue.S3Table(this, 'MyTable', {
1221
+ hasEncryptedData: false,
1222
+ database: myDatabase,
1223
+ columns: [{
1224
+ name: 'col1',
1225
+ type: glue.Schema.STRING,
1226
+ }],
1227
+ dataFormat: glue.DataFormat.JSON,
1228
+ });
1229
+ ```
1230
+
1231
+ Do not set `has_encrypted_data` through the free-form `parameters` map as well - a value
1232
+ there that conflicts with `hasEncryptedData` is rejected at synthesis time.
1149
1233
 
1150
1234
  ## Types
1151
1235
 
@@ -1166,7 +1250,7 @@ new glue.S3Table(this, 'MyTable', {
1166
1250
  type: glue.Schema.map(
1167
1251
  glue.Schema.STRING,
1168
1252
  glue.Schema.TIMESTAMP),
1169
- comment: 'map<string,string>',
1253
+ comment: 'map<string,timestamp>',
1170
1254
  }, {
1171
1255
  name: 'struct_column',
1172
1256
  type: glue.Schema.struct([{
@@ -1182,6 +1266,8 @@ new glue.S3Table(this, 'MyTable', {
1182
1266
  });
1183
1267
  ```
1184
1268
 
1269
+ For a type the `Schema` factories don't model, use `glue.Schema.custom('...')`, which takes the raw Glue input string.
1270
+
1185
1271
  ## Public FAQ
1186
1272
 
1187
1273
  ### What are we launching today?
package/awslint.json CHANGED
@@ -109,6 +109,11 @@
109
109
  "interface-extends-ref:@aws-cdk/aws-glue-alpha.ISecurityConfiguration",
110
110
  "interface-extends-ref:@aws-cdk/aws-glue-alpha.ITable",
111
111
  "interface-extends-ref:@aws-cdk/aws-glue-alpha.IWorkflow",
112
- "prefer-ref-interface:@aws-cdk/aws-glue-alpha.DatabaseProps.catalog"
112
+ "prefer-ref-interface:@aws-cdk/aws-glue-alpha.DatabaseProps.catalog",
113
+ "prefer-ref-interface:@aws-cdk/aws-glue-alpha.ConnectionProps.vpc",
114
+ "prefer-ref-interface:aws-cdk-lib.aws_ec2.ClientVpnEndpointOptions.securityGroups",
115
+ "prefer-ref-interface:aws-cdk-lib.aws_ec2.SubnetSelection.subnets",
116
+ "prefer-ref-interface:aws-cdk-lib.aws_ec2.InterfaceVpcEndpointOptions.securityGroups",
117
+ "no-unused-type:@aws-cdk/aws-glue-alpha.S3EncryptionMode"
113
118
  ]
114
119
  }
package/lib/catalog.js CHANGED
@@ -74,7 +74,7 @@ var CatalogEncryptionMode;
74
74
  * @see https://docs.aws.amazon.com/glue/latest/webapi/API_EncryptionAtRest.html
75
75
  */
76
76
  class DataCatalogEncryptionAtRest {
77
- static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.DataCatalogEncryptionAtRest", version: "2.265.0-alpha.0" };
77
+ static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.DataCatalogEncryptionAtRest", version: "2.267.0-alpha.0" };
78
78
  /**
79
79
  * Disable encryption at rest for the Data Catalog.
80
80
  */
@@ -130,7 +130,7 @@ exports.DataCatalogEncryptionAtRest = DataCatalogEncryptionAtRest;
130
130
  * construction, so a catalog either carries settings or it does not.
131
131
  */
132
132
  class CatalogBase extends core_1.Resource {
133
- static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.CatalogBase", version: "2.265.0-alpha.0" };
133
+ static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.CatalogBase", version: "2.267.0-alpha.0" };
134
134
  _encryptionKey;
135
135
  _connectionPasswordKey;
136
136
  get encryptionKey() {
@@ -224,7 +224,7 @@ let Catalog = (() => {
224
224
  Catalog = _classThis = _classDescriptor.value;
225
225
  if (_metadata) Object.defineProperty(_classThis, Symbol.metadata, { enumerable: true, configurable: true, writable: true, value: _metadata });
226
226
  }
227
- static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.Catalog", version: "2.265.0-alpha.0" };
227
+ static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.Catalog", version: "2.267.0-alpha.0" };
228
228
  /** Uniquely identifies this class. */
229
229
  static PROPERTY_INJECTION_ID = '@aws-cdk.aws-glue-alpha.Catalog';
230
230
  /**
package/lib/code.js CHANGED
@@ -43,7 +43,7 @@ const helpers_internal_1 = require("aws-cdk-lib/core/lib/helpers-internal");
43
43
  * Represents a Glue Job's Code assets (an asset can be a scripts, a jar, a python file or any other file).
44
44
  */
45
45
  class Code {
46
- static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.Code", version: "2.265.0-alpha.0" };
46
+ static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.Code", version: "2.267.0-alpha.0" };
47
47
  /**
48
48
  * Job code as an S3 object.
49
49
  * @param bucket The S3 bucket
@@ -68,7 +68,7 @@ exports.Code = Code;
68
68
  class S3Code extends Code {
69
69
  bucket;
70
70
  key;
71
- static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.S3Code", version: "2.265.0-alpha.0" };
71
+ static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.S3Code", version: "2.267.0-alpha.0" };
72
72
  constructor(bucket, key) {
73
73
  super();
74
74
  this.bucket = bucket;
@@ -91,7 +91,7 @@ exports.S3Code = S3Code;
91
91
  class AssetCode extends Code {
92
92
  path;
93
93
  options;
94
- static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.AssetCode", version: "2.265.0-alpha.0" };
94
+ static [JSII_RTTI_SYMBOL_1] = { fqn: "@aws-cdk/aws-glue-alpha.AssetCode", version: "2.267.0-alpha.0" };
95
95
  asset;
96
96
  /**
97
97
  * @param path The path to the Code file.
@@ -1,4 +1,5 @@
1
1
  import type * as ec2 from 'aws-cdk-lib/aws-ec2';
2
+ import type { ISecretRef } from 'aws-cdk-lib/aws-secretsmanager';
2
3
  import * as cdk from 'aws-cdk-lib/core';
3
4
  import type * as constructs from 'constructs';
4
5
  /**
@@ -219,6 +220,17 @@ export interface ConnectionOptions {
219
220
  readonly properties?: {
220
221
  [key: string]: string;
221
222
  };
223
+ /**
224
+ * A reference to a Secrets Manager secret holding the credentials for this connection.
225
+ *
226
+ * The secret is referenced through the connection's `SECRET_ID` property, so
227
+ * Glue reads the credentials at runtime and the secret value never appears in
228
+ * the synthesized template. Prefer this over placing credentials directly in
229
+ * `properties`. Accepts any `secretsmanager.ISecret`.
230
+ *
231
+ * @default - no secret; any credentials must be supplied via `properties`
232
+ */
233
+ readonly secret?: ISecretRef;
222
234
  /**
223
235
  * A list of criteria that can be used in selecting this connection.
224
236
  * This is useful for filtering the results of https://awscli.amazonaws.com/v2/documentation/api/latest/reference/glue/get-connections.html
@@ -232,9 +244,32 @@ export interface ConnectionOptions {
232
244
  readonly securityGroups?: ec2.ISecurityGroup[];
233
245
  /**
234
246
  * The VPC subnet to connect to resources within a VPC. See more at https://docs.aws.amazon.com/glue/latest/dg/start-connecting.html.
235
- * @default no subnet
247
+ *
248
+ * Mutually exclusive with `vpc`: provide `subnet` to pin the connection to a
249
+ * specific subnet, or provide `vpc` (optionally with `vpcSubnets`) to let the
250
+ * CDK select one for you.
251
+ *
252
+ * @default - no subnet, unless `vpc` is provided
236
253
  */
237
254
  readonly subnet?: ec2.ISubnet;
255
+ /**
256
+ * The VPC to connect to resources within. When provided, the CDK selects a
257
+ * subnet from this VPC using `vpcSubnets`. A Glue connection targets a single
258
+ * subnet, so the first subnet of the selection is used.
259
+ *
260
+ * Mutually exclusive with `subnet`.
261
+ *
262
+ * @default - no VPC, the subnet is taken from `subnet` if provided
263
+ */
264
+ readonly vpc?: ec2.IVpc;
265
+ /**
266
+ * Which subnets of `vpc` to select the connection subnet from. Only used when
267
+ * `vpc` is provided. Since a Glue connection targets a single subnet, the
268
+ * first subnet of the selection is used.
269
+ *
270
+ * @default - private subnets
271
+ */
272
+ readonly vpcSubnets?: ec2.SubnetSelection;
238
273
  }
239
274
  /**
240
275
  * Construction properties for `Connection`
@@ -271,6 +306,11 @@ export declare class Connection extends cdk.Resource implements IConnection {
271
306
  private readonly properties;
272
307
  private readonly resource;
273
308
  constructor(scope: constructs.Construct, id: string, props: ConnectionProps);
309
+ /**
310
+ * Determines the single subnet the connection should target, either from an
311
+ * explicit `subnet` or by selecting one from `vpc`.
312
+ */
313
+ private resolveSubnet;
274
314
  /**
275
315
  * The name of the connection
276
316
  */