@go-to-k/cdkd 0.283.8 → 0.283.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -56,219 +56,560 @@ var __exportAll = (all, no_symbols) => {
56
56
  };
57
57
 
58
58
  //#endregion
59
- //#region src/utils/error-handler.ts
60
- /**
61
- * Base error class for cdkd
62
- */
63
- var CdkdError = class CdkdError extends Error {
64
- code;
65
- cause;
66
- constructor(message, code, cause) {
67
- super(message);
68
- this.code = code;
69
- this.cause = cause;
70
- this.name = "CdkdError";
71
- Object.setPrototypeOf(this, CdkdError.prototype);
72
- }
73
- };
59
+ //#region src/deployment/retryable-errors.ts
74
60
  /**
75
- * State management errors
61
+ * The **IAM-propagation** subset of {@link RETRYABLE_ERROR_MESSAGE_PATTERNS}:
62
+ * an AWS service rejecting a call because a just-created IAM entity (role,
63
+ * trust policy, inline policy, instance profile, principal) has not propagated
64
+ * to that service's authorization layer yet.
65
+ *
66
+ * Kept as its own array — and composed back into the full transient table
67
+ * below — so there is exactly ONE list per pattern (no parallel classifier to
68
+ * drift). It exists because this class has a materially different RECOVERY
69
+ * SHAPE from the other transient errors: it resolves in single-digit seconds,
70
+ * so `withRetry` polls it on a dense sub-second schedule instead of the
71
+ * generic 1s/2s/4s/8s exponential backoff (which is right for throttling and
72
+ * for long resource-state transitions, and wrong here — see
73
+ * {@link file://../deployment/retry.ts}).
74
+ *
75
+ * When adding a new pattern: put it here if the fix is "wait a moment and ask
76
+ * IAM again", and in `OTHER_TRANSIENT_ERROR_MESSAGE_PATTERNS` otherwise. A
77
+ * misfiled entry only changes the retry CADENCE, never whether the error is
78
+ * retryable at all.
76
79
  */
77
- var StateError = class StateError extends CdkdError {
78
- constructor(message, cause) {
79
- super(message, "STATE_ERROR", cause);
80
- this.name = "StateError";
81
- Object.setPrototypeOf(this, StateError.prototype);
82
- }
83
- };
80
+ const IAM_PROPAGATION_ERROR_MESSAGE_PATTERNS = [
81
+ "cannot be assumed",
82
+ "Firehose is unable to assume role",
83
+ "is unable to assume provided role",
84
+ "is unable to assume the role",
85
+ "security token included in the request is invalid. (Service:",
86
+ "role defined for the function",
87
+ "not authorized to perform",
88
+ "execution role",
89
+ "trust policy",
90
+ "Role validation failed",
91
+ "does not have required permissions",
92
+ "Trusted Entity",
93
+ "Invalid principal in policy",
94
+ "Verify in IAM that the role has adequate trust relationships",
95
+ "The user with name",
96
+ "Policy Error: PrincipalNotFound",
97
+ "Invalid value for the parameter Policy",
98
+ "required permissions for: ENHANCED_MONITORING",
99
+ "Caught ServiceAccessDeniedException",
100
+ "permissions required to assume the role",
101
+ "authorized to assume the provided role",
102
+ "Cannot access stream",
103
+ "Please ensure the role can perform",
104
+ "KMS key is invalid for CreateGrant",
105
+ "Policy contains a statement with one or more invalid principals",
106
+ "Invalid IAM Instance Profile",
107
+ "Invalid InstanceProfile",
108
+ "Failed to authorize instance profile",
109
+ "is not a valid role to allow SNS"
110
+ ];
84
111
  /**
85
- * Lock acquisition errors
112
+ * The NON-IAM-propagation half of {@link RETRYABLE_ERROR_MESSAGE_PATTERNS}:
113
+ * transient failures whose recovery window is either long (SQS's 60s same-name
114
+ * cooldown, a resource still leaving a Pending/Creating state) or genuinely
115
+ * load-related (throttling), where hammering AWS with dense retries is harmful
116
+ * and exponential backoff is the correct shape.
86
117
  */
87
- var LockError = class LockError extends CdkdError {
88
- constructor(message, cause) {
89
- super(message, "LOCK_ERROR", cause);
90
- this.name = "LockError";
91
- Object.setPrototypeOf(this, LockError.prototype);
92
- }
93
- };
118
+ const OTHER_TRANSIENT_ERROR_MESSAGE_PATTERNS = [
119
+ "currently in the following state: Pending",
120
+ "has dependencies and cannot be deleted",
121
+ "can't be deleted since it has",
122
+ "DependencyViolation",
123
+ "does not exist",
124
+ "Schema is currently being altered",
125
+ "conflicting conditional operation",
126
+ "scheduled for deletion",
127
+ "Could not deliver test message",
128
+ "wait 60 seconds",
129
+ "concurrent update operation",
130
+ "because it is in use",
131
+ "Rate exceeded",
132
+ "is marked disabled for mutation",
133
+ "There is an operation running on the Cluster",
134
+ "Unable to complete operation due to concurrent modification"
135
+ ];
94
136
  /**
95
- * Synthesis errors
137
+ * Patterns that mark an AWS error as a transient/retryable failure.
138
+ * Each entry is a substring match against the error message; all of these
139
+ * are situations where the same call typically succeeds after a short delay
140
+ * because of eventual consistency or just-created-dependency propagation.
141
+ *
142
+ * Composed from the two halves above so retryability has ONE source of truth
143
+ * while `withRetry` can still pick a per-class backoff cadence.
96
144
  */
97
- var SynthesisError = class SynthesisError extends CdkdError {
98
- constructor(message, cause) {
99
- super(message, "SYNTHESIS_ERROR", cause);
100
- this.name = "SynthesisError";
101
- Object.setPrototypeOf(this, SynthesisError.prototype);
102
- }
103
- };
145
+ const RETRYABLE_ERROR_MESSAGE_PATTERNS = [...IAM_PROPAGATION_ERROR_MESSAGE_PATTERNS, ...OTHER_TRANSIENT_ERROR_MESSAGE_PATTERNS];
104
146
  /**
105
- * Control-flow signal: the user declined a pre-provisioning confirmation
106
- * prompt, so the deploy must unwind WITHOUT being reported as a failure.
107
- *
108
- * Raised from `DeployEngineOptions.onCurrentStateLoaded` (the post-lock
109
- * gate the `--prefix-user-supplied-names` migration check runs in). The
110
- * engine does not catch it — it propagates out of `deploy()` through the
111
- * usual `finally`, which releases the lock and stops the renderer — and the
112
- * deploy CLI catches it and returns quietly instead of logging an error or
113
- * recording a FAILED run event. Nothing has been provisioned at that point.
147
+ * HTTP status codes that always indicate a transient failure worth retrying.
148
+ * 429 = Too Many Requests (throttle), 503 = Service Unavailable.
114
149
  */
115
- var DeployCancelledError = class DeployCancelledError extends CdkdError {
116
- constructor(message = "Deployment cancelled by user") {
117
- super(message, "DEPLOY_CANCELLED");
118
- this.name = "DeployCancelledError";
119
- Object.setPrototypeOf(this, DeployCancelledError.prototype);
120
- }
121
- };
150
+ const RETRYABLE_HTTP_STATUS_CODES = /* @__PURE__ */ new Set([429, 503]);
122
151
  /**
123
- * Asset errors
152
+ * AWS SDK v3 canonical throttling error names. Mirrors
153
+ * `@aws-sdk/service-error-classification`'s `THROTTLING_ERROR_CODES` — any
154
+ * error (or wrapped cause) whose `name` is one of these is a transient rate-
155
+ * limit rejection worth retrying with backoff. Detecting by NAME is more
156
+ * robust than by HTTP status because most AWS throttles surface as HTTP 400
157
+ * (not 429) with the throttling signal carried only in the error code / name
158
+ * (e.g. SSM `ThrottlingException` for the `Rate exceeded` message).
124
159
  */
125
- var AssetError = class AssetError extends CdkdError {
126
- constructor(message, cause) {
127
- super(message, "ASSET_ERROR", cause);
128
- this.name = "AssetError";
129
- Object.setPrototypeOf(this, AssetError.prototype);
130
- }
131
- };
160
+ const THROTTLING_ERROR_NAMES = /* @__PURE__ */ new Set([
161
+ "BandwidthLimitExceeded",
162
+ "EC2ThrottledException",
163
+ "LimitExceededException",
164
+ "PriorRequestNotComplete",
165
+ "ProvisionedThroughputExceededException",
166
+ "RequestLimitExceeded",
167
+ "RequestThrottled",
168
+ "RequestThrottledException",
169
+ "SlowDown",
170
+ "ThrottledException",
171
+ "Throttling",
172
+ "ThrottlingException",
173
+ "TooManyRequestsException",
174
+ "TransactionInProgressException"
175
+ ]);
132
176
  /**
133
- * Local-invoke `docker build` failures.
177
+ * Marker for an error cdkd raised as a DELIBERATE refusal rather than as a
178
+ * relayed AWS failure (issue [#1778](https://github.com/go-to-k/cdkd/issues/1778)).
134
179
  *
135
- * Surfaces the stderr captured from `docker build` so the user can
136
- * re-run the same command directly to debug Dockerfile syntax errors
137
- * or missing build context. Used by `src/local/docker-image-builder.ts`
138
- * (PR 5) for container Lambdas; the parallel `AssetError` covers the
139
- * `cdkd publish-assets` / `cdkd deploy` build path. Kept distinct from
140
- * `AssetError` so `cdkd local invoke` failures don't show up under the
141
- * "asset" error class.
180
+ * Every classifier below is SUBSTRING-based, which is the right shape for
181
+ * relaying a vendor's message and the wrong one for cdkd's own prose: a
182
+ * refusal message is assembled from values cdkd does not control — a provider
183
+ * `reason`, a state-borne physicalId, a template logical id — and any of them
184
+ * can happen to contain a retryable pattern. Measured: a resource named
185
+ * `MyDependencyViolationSub` puts `DependencyViolation` in the message, so a
186
+ * deterministic refusal was classified transient and burned the whole backoff
187
+ * schedule before failing exactly as it would have immediately. Keeping the
188
+ * offending values OUT of the message narrows that surface but cannot close
189
+ * it, because a message with no identifiers at all is not diagnosable.
190
+ *
191
+ * A marker inverts the burden: the raiser STATES that the error is terminal,
192
+ * so no wording can make it retryable. Deliberately a `Symbol.for` key —
193
+ * global-registry symbols survive a duplicated module instance (dual
194
+ * bundling), where a module-local symbol would silently stop matching — and
195
+ * non-enumerable, so it cannot leak into a serialized error payload.
142
196
  */
143
- var LocalInvokeBuildError = class LocalInvokeBuildError extends CdkdError {
144
- constructor(message, cause) {
145
- super(message, "LOCAL_INVOKE_BUILD_ERROR", cause);
146
- this.name = "LocalInvokeBuildError";
147
- Object.setPrototypeOf(this, LocalInvokeBuildError.prototype);
148
- }
149
- };
197
+ const NON_RETRYABLE_MARKER = Symbol.for("cdkd.nonRetryable");
150
198
  /**
151
- * Resource provisioning errors
199
+ * Mark a cdkd-authored refusal as terminal and return it, for
200
+ * `throw markNonRetryable(new ProvisioningError(...))`.
201
+ *
202
+ * Reach for it when the error means "this cannot succeed on a retry" as a
203
+ * matter of cdkd's own logic — NOT for a relayed AWS failure, whose
204
+ * retryability is the classifiers' business.
205
+ *
206
+ * The known live instance this JSDoc used to flag as uncovered —
207
+ * `ResourceUpdateNotSupportedError` (`src/utils/error-handler.ts`), which
208
+ * interpolates the logical id and is thrown by ~20 providers from inside the
209
+ * retried `update()` in `deploy-engine.ts` — IS covered as of issue
210
+ * [#1838](https://github.com/go-to-k/cdkd/issues/1838): it marks itself in its
211
+ * CONSTRUCTOR, so every construction is terminal and no provider throw site
212
+ * has to remember. That is the shape to prefer for a whole error CLASS that is
213
+ * always a refusal; mark at the `throw` (as the SNS abort below does) only
214
+ * when the class is retryable in general and this one raising of it is not.
152
215
  */
153
- var ProvisioningError = class ProvisioningError extends CdkdError {
154
- resourceType;
155
- logicalId;
156
- physicalId;
157
- constructor(message, resourceType, logicalId, physicalId, cause) {
158
- super(message, "PROVISIONING_ERROR", cause);
159
- this.resourceType = resourceType;
160
- this.logicalId = logicalId;
161
- this.physicalId = physicalId;
162
- this.name = "ProvisioningError";
163
- Object.setPrototypeOf(this, ProvisioningError.prototype);
164
- }
165
- };
216
+ function markNonRetryable(error) {
217
+ if (!Object.isExtensible(error)) return error;
218
+ Object.defineProperty(error, NON_RETRYABLE_MARKER, {
219
+ value: true,
220
+ enumerable: false,
221
+ configurable: true,
222
+ writable: false
223
+ });
224
+ return error;
225
+ }
166
226
  /**
167
- * Resource provisioning timeout errors (per-resource wall-clock deadline).
227
+ * True when the error, or anything in its bounded `.cause` chain, was marked
228
+ * by {@link markNonRetryable}.
168
229
  *
169
- * Thrown by `withResourceDeadline` when a single CREATE / UPDATE / DELETE
170
- * operation exceeds the user-configured `--resource-timeout`. The deploy
171
- * engine catches this, wraps it in {@link ProvisioningError}, and lets the
172
- * existing failure path (interrupt siblings → pre-rollback save → rollback
173
- * unless `--no-rollback`) take over.
230
+ * The chain walk mirrors {@link isThrottlingError}'s: cdkd wraps errors, so a
231
+ * marked refusal can end up one or more links deep, and a marker that stopped
232
+ * counting after a single wrap would be a fence that quietly falls open.
174
233
  *
175
- * The message intentionally names the resource, type, region, elapsed time
176
- * and operation, plus how to override the default. Long-running providers
177
- * (e.g. Custom Resource: 1h polling cap) self-report their needed budget
178
- * via `getMinResourceTimeoutMs()`, so the user only needs a per-type
179
- * override (`--resource-timeout TYPE=DURATION`) when they want to bump a
180
- * specific non-self-reporting type or shorten a self-reported one.
181
- */
182
- var ResourceTimeoutError = class ResourceTimeoutError extends CdkdError {
183
- logicalId;
184
- resourceType;
185
- region;
186
- elapsedMs;
187
- operation;
188
- timeoutMs;
189
- constructor(logicalId, resourceType, region, elapsedMs, operation, timeoutMs) {
190
- const elapsedLabel = formatDuration(elapsedMs);
191
- const timeoutLabel = formatDuration(timeoutMs);
192
- super(`Resource ${logicalId} (${resourceType}) in ${region} timed out after ${timeoutLabel} during ${operation} (elapsed ${elapsedLabel}).\nThis may indicate a stuck Cloud Control polling loop, hung Custom Resource, or
193
- slow ENI provisioning. Re-run with --resource-timeout ${resourceType}=<DURATION>\nto bump the budget for this resource type only, or --verbose to see the
194
- underlying provider activity.`, "RESOURCE_TIMEOUT");
195
- this.logicalId = logicalId;
196
- this.resourceType = resourceType;
197
- this.region = region;
198
- this.elapsedMs = elapsedMs;
199
- this.operation = operation;
200
- this.timeoutMs = timeoutMs;
201
- this.name = "ResourceTimeoutError";
202
- Object.setPrototypeOf(this, ResourceTimeoutError.prototype);
234
+ * That reach is DIRECTIONAL, and the upward direction is a hazard worth
235
+ * stating. Downward — a marked refusal wrapped by an outer error — is the
236
+ * intended case and stays terminal. UPWARD is the inverse: wrapping a marked
237
+ * refusal as the `cause` of a genuinely RETRYABLE outer error
238
+ * (`new Error(msg, { cause: markedRefusal })`) makes the outer error terminal
239
+ * too, because this walk finds the marker on the cause. Unconstructible today
240
+ * (`ProvisioningError` is built with no `cause` at the one marking site), and
241
+ * the failure is fail-fast rather than silent, but a future wrapper that
242
+ * carries a marked cause into a transient error would stop retrying something
243
+ * that should retry. Strip or re-raise the cause there rather than nesting it.
244
+ */
245
+ function isMarkedNonRetryable(error) {
246
+ let current = error;
247
+ for (let depth = 0; depth < 5 && current != null; depth++) {
248
+ if (typeof current === "object" || typeof current === "function") {
249
+ if (current[NON_RETRYABLE_MARKER] === true) return true;
250
+ }
251
+ current = current.cause;
203
252
  }
204
- };
253
+ return false;
254
+ }
205
255
  /**
206
- * Format a duration in milliseconds as a short human-readable label
207
- * (`30m`, `1h30m`, `45s`). Used by {@link ResourceTimeoutError} so the
208
- * error message stays compact.
256
+ * Walk the error + its `.cause` chain (bounded) looking for a rate-limit
257
+ * signal — either an AWS SDK v3 throttling error `name`
258
+ * ({@link THROTTLING_ERROR_NAMES}) or a retryable HTTP status
259
+ * ({@link RETRYABLE_HTTP_STATUS_CODES}) on `$metadata`.
260
+ *
261
+ * cdkd wraps the original AWS error in a `ProvisioningError`, so the signal is
262
+ * typically one cause-link deep; the bounded walk also tolerates SDK errors
263
+ * that nest a `$response`/cause without exploding on a cyclic chain.
264
+ *
265
+ * BOTH signals are checked at EVERY depth. An earlier version checked the name
266
+ * to depth 5 but the HTTP status only at depths 0 and 1, so a 429 nested two
267
+ * links deep was missed.
209
268
  */
210
- function formatDuration(ms) {
211
- if (ms < 6e4) return `${Math.round(ms / 1e3)}s`;
212
- const totalMinutes = Math.round(ms / 6e4);
213
- if (totalMinutes < 60) return `${totalMinutes}m`;
214
- const hours = Math.floor(totalMinutes / 60);
215
- const minutes = totalMinutes % 60;
216
- return minutes === 0 ? `${hours}h` : `${hours}h${minutes}m`;
269
+ function isThrottlingError(error) {
270
+ let current = error;
271
+ for (let depth = 0; depth < 5 && current != null; depth++) {
272
+ const name = current.name;
273
+ if (typeof name === "string" && THROTTLING_ERROR_NAMES.has(name)) return true;
274
+ const status = current.$metadata?.httpStatusCode;
275
+ if (status !== void 0 && RETRYABLE_HTTP_STATUS_CODES.has(status)) return true;
276
+ current = current.cause;
277
+ }
278
+ return false;
217
279
  }
218
280
  /**
219
- * A DELIBERATE refusal to resolve an intrinsic function, as opposed to a
220
- * "the referenced thing does not exist" miss (issue
221
- * [#1740](https://github.com/go-to-k/cdkd/issues/1740)).
281
+ * Determine whether an AWS error should be retried.
222
282
  *
223
- * The distinction exists for exactly one consumer: `Fn::Sub`'s variable
224
- * resolution, which speculatively tries `Ref` and then `Fn::GetAtt` and keeps
225
- * the raw `${...}` placeholder when neither resolves. That warn-and-keep is the
226
- * long-standing, deliberate behavior for a genuinely unknown variable — but a
227
- * bare `catch` around it also swallowed every REFUSAL the resolver raises on
228
- * purpose (`guardedPhysicalIdFallback`'s ARN / URL shape hard-fail, the
229
- * `--strict-getatt` rejection, `rejectPlaceholderArnAttribute`), so a template
230
- * that hard-fails when the reference sits in a resource property silently
231
- * degraded to shipping a literal `${Resource.Attribute}` to AWS when the
232
- * IDENTICAL reference was written inside an `Fn::Sub`.
283
+ * Checks (in order):
284
+ * 0. {@link isMarkedNonRetryable} — a cdkd-authored refusal is terminal by
285
+ * declaration, ahead of every message / name heuristic below. FIRST on
286
+ * purpose: the marker states the error cannot succeed on a retry, so
287
+ * nothing a later check reads out of the message can overturn it.
288
+ * 1. Rate-limit signal on the error or any wrapped cause — throttling error
289
+ * `name` or retryable HTTP status (most AWS throttles are HTTP 400, not
290
+ * 429, so the name check carries most of the weight). See
291
+ * {@link isThrottlingError}.
292
+ * 2. Substring match against {@link RETRYABLE_ERROR_MESSAGE_PATTERNS}
293
+ */
294
+ function isRetryableTransientError(error, message) {
295
+ if (isMarkedNonRetryable(error)) return false;
296
+ if (isThrottlingError(error)) return true;
297
+ return RETRYABLE_ERROR_MESSAGE_PATTERNS.some((p) => message.includes(p));
298
+ }
299
+ /**
300
+ * True when the message is a just-created-IAM-entity propagation rejection
301
+ * ({@link IAM_PROPAGATION_ERROR_MESSAGE_PATTERNS}).
233
302
  *
234
- * Throwing this class rather than a bare `Error` is what lets that catch
235
- * re-raise a refusal (carrying its own message and remedy) while leaving the
236
- * not-found path on warn-and-keep. Nothing else branches on it.
303
+ * This does NOT decide retryability — every pattern it matches is already in
304
+ * {@link RETRYABLE_ERROR_MESSAGE_PATTERNS}. It only selects the retry CADENCE:
305
+ * `withRetry` polls this class densely (sub-second initial delay, low cap)
306
+ * because IAM propagation resolves in single-digit seconds, whereas the
307
+ * generic exponential schedule is tuned for throttling and long resource-state
308
+ * transitions.
309
+ *
310
+ * Deliberately message-only (no error-object inspection): the propagation
311
+ * signal is always carried in the vendor's message text, and cdkd wraps the
312
+ * original error in a `ProvisioningError` that preserves it.
237
313
  */
238
- var IntrinsicResolutionRefusalError = class IntrinsicResolutionRefusalError extends CdkdError {
314
+ function isIamPropagationError(message) {
315
+ return IAM_PROPAGATION_ERROR_MESSAGE_PATTERNS.some((p) => message.includes(p));
316
+ }
317
+ /**
318
+ * Match the "already exists" name-collision signature raised when a create
319
+ * targets a physical name still held by another resource (or by the same
320
+ * name's not-yet-released tombstone after an async delete).
321
+ *
322
+ * Deliberately NOT part of {@link RETRYABLE_ERROR_MESSAGE_PATTERNS}: a name
323
+ * collision is only worth retrying at the specific re-create sites that just
324
+ * deleted the old holder (the deploy engine's --replace delete-first fallback
325
+ * and the rollback executor's reverse-replacement) — everywhere else it is a
326
+ * genuine conflict that must fail fast. Shared by those sites' collision
327
+ * detection + retry filters so a signature extension lands in one place.
328
+ *
329
+ * The optional `s` is load-bearing, not defensive spelling (issue #1625):
330
+ * Lambda's `CreateFunction` raises `ResourceConflictException: Function
331
+ * already exist: <name>` — SINGULAR — so the `already exists` form missed it
332
+ * entirely and NO Lambda function could take the collision path. The
333
+ * consequence was not a cosmetic message: a property-driven replacement of a
334
+ * Lambda (dropping `DurableConfig`, changing `TenancyConfig`) create-firsts
335
+ * into its own still-live name, the raw `ResourceConflictException` escaped
336
+ * instead of the actionable `NAMED_REPLACEMENT_COLLISION` error, and
337
+ * `cdkd deploy --replace`'s delete-first fallback never fired — so the
338
+ * replacement was unperformable by any flag. Verified against real AWS
339
+ * (us-east-1, 2026-08-12) by creating one function name twice.
340
+ *
341
+ * Two fences keep the widened arm from crediting a NON-collision, which
342
+ * matters because the sites that consult it react DESTRUCTIVELY (the
343
+ * `--replace` delete-first fallback deletes the old resource):
344
+ * - `\b` after `exists?` rejects a participle ("already existed as a draft");
345
+ * - the lookbehind rejects a NEGATED or MODAL phrase — "the bucket does NOT
346
+ * already exist", "the destination bucket MUST already exist" — which the
347
+ * bare pattern matched. The modal form is the one that bites: a create
348
+ * rejected for a missing PREREQUISITE would be reported to the user as a
349
+ * name collision pointing at `--replace`, and following that advice
350
+ * deletes the live old resource before the re-create fails again for the
351
+ * same reason, leaving it absent from AWS with state still recording it.
352
+ * The error-CODE arm stays exact (`AlreadyExists`): the singular
353
+ * `AlreadyExist` is not an AWS code spelling, and loosening it would match
354
+ * inside unrelated identifiers.
355
+ *
356
+ * Classification stays MESSAGE-based rather than moving to the exception
357
+ * NAME, and that is load-bearing here rather than inherited: Lambda raises
358
+ * `ResourceConflictException` for a function in a PENDING state too (see
359
+ * `lambda-function-provider.ts`), so keying on the name would classify a
360
+ * transient state conflict as a collision and delete a live function under
361
+ * `--replace`.
362
+ */
363
+ function isNameCollisionError(message) {
364
+ return /(?<!\b(?:must|not|should|may|cannot)\s)already exists?\b/i.test(message) || message.includes("AlreadyExists");
365
+ }
366
+ /**
367
+ * Match the SQS same-name re-creation cooldown: after `DeleteQueue`, creating
368
+ * a queue with the SAME name inside ~60s fails with
369
+ * `AWS.SimpleQueueService.QueueDeletedRecently` ("You must wait 60 seconds
370
+ * after deleting a queue before you can create another with the same name").
371
+ *
372
+ * The generic transient table above already carries 'wait 60 seconds' for
373
+ * plain CREATEs (rapid destroy → redeploy loops), but the delete-then-re-create
374
+ * sites (the deploy engine's --replace delete-first fallback and the rollback
375
+ * executor's reverse-replacement) override the retry filter with
376
+ * {@link isNameCollisionError}, which this signature does NOT match — so a
377
+ * replacement revert used to fail fast mid-flight with the resource absent
378
+ * from both AWS and state (issue #1206). Those sites now OR this matcher into
379
+ * their retry filter, with a schedule long enough to cover the 60s window.
380
+ *
381
+ * Kept separate from {@link isNameCollisionError} on purpose: a cooldown at a
382
+ * create-first site must NOT be treated as a collision (deleting the new
383
+ * resource would not release the cooldown on the old name).
384
+ */
385
+ function isNameCooldownError(message) {
386
+ return message.includes("QueueDeletedRecently") || message.includes("wait 60 seconds");
387
+ }
388
+ /**
389
+ * Retry filter for the delete-then-re-create sites: the old name holder was
390
+ * just deleted, so both the late name release ("already exists" from an async
391
+ * delete) and the SQS 60s name cooldown are worth waiting out. Pair with a
392
+ * schedule that covers the full cooldown window (maxRetries 8, delays
393
+ * 2s/4s/8s then capped at 10s ≈ 64s total sleep).
394
+ */
395
+ function isRecreateRetryableError(message) {
396
+ return isNameCollisionError(message) || isNameCooldownError(message);
397
+ }
398
+
399
+ //#endregion
400
+ //#region src/utils/error-handler.ts
401
+ /**
402
+ * Base error class for cdkd
403
+ */
404
+ var CdkdError = class CdkdError extends Error {
405
+ code;
406
+ cause;
407
+ constructor(message, code, cause) {
408
+ super(message);
409
+ this.code = code;
410
+ this.cause = cause;
411
+ this.name = "CdkdError";
412
+ Object.setPrototypeOf(this, CdkdError.prototype);
413
+ }
414
+ };
415
+ /**
416
+ * State management errors
417
+ */
418
+ var StateError = class StateError extends CdkdError {
239
419
  constructor(message, cause) {
240
- super(message, "INTRINSIC_RESOLUTION_REFUSAL", cause);
241
- this.name = "IntrinsicResolutionRefusalError";
242
- Object.setPrototypeOf(this, IntrinsicResolutionRefusalError.prototype);
420
+ super(message, "STATE_ERROR", cause);
421
+ this.name = "StateError";
422
+ Object.setPrototypeOf(this, StateError.prototype);
243
423
  }
244
424
  };
245
425
  /**
246
- * Dependency resolution errors
426
+ * Lock acquisition errors
247
427
  */
248
- var DependencyError = class DependencyError extends CdkdError {
428
+ var LockError = class LockError extends CdkdError {
249
429
  constructor(message, cause) {
250
- super(message, "DEPENDENCY_ERROR", cause);
251
- this.name = "DependencyError";
252
- Object.setPrototypeOf(this, DependencyError.prototype);
430
+ super(message, "LOCK_ERROR", cause);
431
+ this.name = "LockError";
432
+ Object.setPrototypeOf(this, LockError.prototype);
253
433
  }
254
434
  };
255
435
  /**
256
- * Configuration errors
436
+ * Synthesis errors
257
437
  */
258
- var ConfigError = class ConfigError extends CdkdError {
438
+ var SynthesisError = class SynthesisError extends CdkdError {
259
439
  constructor(message, cause) {
260
- super(message, "CONFIG_ERROR", cause);
261
- this.name = "ConfigError";
262
- Object.setPrototypeOf(this, ConfigError.prototype);
440
+ super(message, "SYNTHESIS_ERROR", cause);
441
+ this.name = "SynthesisError";
442
+ Object.setPrototypeOf(this, SynthesisError.prototype);
263
443
  }
264
444
  };
265
445
  /**
266
- * Signals a partial-failure outcome that should map to exit code 2 (not 1).
446
+ * Control-flow signal: the user declined a pre-provisioning confirmation
447
+ * prompt, so the deploy must unwind WITHOUT being reported as a failure.
267
448
  *
268
- * Used by `cdkd destroy` and `cdkd state destroy` when one or more
269
- * per-resource deletes failed but the overall command finished its work
270
- * (state.json is preserved, the rest of the stack was deleted, and the
271
- * user can re-run to clean up the remaining resources).
449
+ * Raised from `DeployEngineOptions.onCurrentStateLoaded` (the post-lock
450
+ * gate the `--prefix-user-supplied-names` migration check runs in). The
451
+ * engine does not catch it — it propagates out of `deploy()` through the
452
+ * usual `finally`, which releases the lock and stops the renderer — and the
453
+ * deploy CLI catches it and returns quietly instead of logging an error or
454
+ * recording a FAILED run event. Nothing has been provisioned at that point.
455
+ */
456
+ var DeployCancelledError = class DeployCancelledError extends CdkdError {
457
+ constructor(message = "Deployment cancelled by user") {
458
+ super(message, "DEPLOY_CANCELLED");
459
+ this.name = "DeployCancelledError";
460
+ Object.setPrototypeOf(this, DeployCancelledError.prototype);
461
+ }
462
+ };
463
+ /**
464
+ * Asset errors
465
+ */
466
+ var AssetError = class AssetError extends CdkdError {
467
+ constructor(message, cause) {
468
+ super(message, "ASSET_ERROR", cause);
469
+ this.name = "AssetError";
470
+ Object.setPrototypeOf(this, AssetError.prototype);
471
+ }
472
+ };
473
+ /**
474
+ * Local-invoke `docker build` failures.
475
+ *
476
+ * Surfaces the stderr captured from `docker build` so the user can
477
+ * re-run the same command directly to debug Dockerfile syntax errors
478
+ * or missing build context. Used by `src/local/docker-image-builder.ts`
479
+ * (PR 5) for container Lambdas; the parallel `AssetError` covers the
480
+ * `cdkd publish-assets` / `cdkd deploy` build path. Kept distinct from
481
+ * `AssetError` so `cdkd local invoke` failures don't show up under the
482
+ * "asset" error class.
483
+ */
484
+ var LocalInvokeBuildError = class LocalInvokeBuildError extends CdkdError {
485
+ constructor(message, cause) {
486
+ super(message, "LOCAL_INVOKE_BUILD_ERROR", cause);
487
+ this.name = "LocalInvokeBuildError";
488
+ Object.setPrototypeOf(this, LocalInvokeBuildError.prototype);
489
+ }
490
+ };
491
+ /**
492
+ * Resource provisioning errors
493
+ */
494
+ var ProvisioningError = class ProvisioningError extends CdkdError {
495
+ resourceType;
496
+ logicalId;
497
+ physicalId;
498
+ constructor(message, resourceType, logicalId, physicalId, cause) {
499
+ super(message, "PROVISIONING_ERROR", cause);
500
+ this.resourceType = resourceType;
501
+ this.logicalId = logicalId;
502
+ this.physicalId = physicalId;
503
+ this.name = "ProvisioningError";
504
+ Object.setPrototypeOf(this, ProvisioningError.prototype);
505
+ }
506
+ };
507
+ /**
508
+ * Resource provisioning timeout errors (per-resource wall-clock deadline).
509
+ *
510
+ * Thrown by `withResourceDeadline` when a single CREATE / UPDATE / DELETE
511
+ * operation exceeds the user-configured `--resource-timeout`. The deploy
512
+ * engine catches this, wraps it in {@link ProvisioningError}, and lets the
513
+ * existing failure path (interrupt siblings → pre-rollback save → rollback
514
+ * unless `--no-rollback`) take over.
515
+ *
516
+ * The message intentionally names the resource, type, region, elapsed time
517
+ * and operation, plus how to override the default. Long-running providers
518
+ * (e.g. Custom Resource: 1h polling cap) self-report their needed budget
519
+ * via `getMinResourceTimeoutMs()`, so the user only needs a per-type
520
+ * override (`--resource-timeout TYPE=DURATION`) when they want to bump a
521
+ * specific non-self-reporting type or shorten a self-reported one.
522
+ */
523
+ var ResourceTimeoutError = class ResourceTimeoutError extends CdkdError {
524
+ logicalId;
525
+ resourceType;
526
+ region;
527
+ elapsedMs;
528
+ operation;
529
+ timeoutMs;
530
+ constructor(logicalId, resourceType, region, elapsedMs, operation, timeoutMs) {
531
+ const elapsedLabel = formatDuration(elapsedMs);
532
+ const timeoutLabel = formatDuration(timeoutMs);
533
+ super(`Resource ${logicalId} (${resourceType}) in ${region} timed out after ${timeoutLabel} during ${operation} (elapsed ${elapsedLabel}).\nThis may indicate a stuck Cloud Control polling loop, hung Custom Resource, or
534
+ slow ENI provisioning. Re-run with --resource-timeout ${resourceType}=<DURATION>\nto bump the budget for this resource type only, or --verbose to see the
535
+ underlying provider activity.`, "RESOURCE_TIMEOUT");
536
+ this.logicalId = logicalId;
537
+ this.resourceType = resourceType;
538
+ this.region = region;
539
+ this.elapsedMs = elapsedMs;
540
+ this.operation = operation;
541
+ this.timeoutMs = timeoutMs;
542
+ this.name = "ResourceTimeoutError";
543
+ Object.setPrototypeOf(this, ResourceTimeoutError.prototype);
544
+ }
545
+ };
546
+ /**
547
+ * Format a duration in milliseconds as a short human-readable label
548
+ * (`30m`, `1h30m`, `45s`). Used by {@link ResourceTimeoutError} so the
549
+ * error message stays compact.
550
+ */
551
+ function formatDuration(ms) {
552
+ if (ms < 6e4) return `${Math.round(ms / 1e3)}s`;
553
+ const totalMinutes = Math.round(ms / 6e4);
554
+ if (totalMinutes < 60) return `${totalMinutes}m`;
555
+ const hours = Math.floor(totalMinutes / 60);
556
+ const minutes = totalMinutes % 60;
557
+ return minutes === 0 ? `${hours}h` : `${hours}h${minutes}m`;
558
+ }
559
+ /**
560
+ * A DELIBERATE refusal to resolve an intrinsic function, as opposed to a
561
+ * "the referenced thing does not exist" miss (issue
562
+ * [#1740](https://github.com/go-to-k/cdkd/issues/1740)).
563
+ *
564
+ * The distinction exists for exactly one consumer: `Fn::Sub`'s variable
565
+ * resolution, which speculatively tries `Ref` and then `Fn::GetAtt` and keeps
566
+ * the raw `${...}` placeholder when neither resolves. That warn-and-keep is the
567
+ * long-standing, deliberate behavior for a genuinely unknown variable — but a
568
+ * bare `catch` around it also swallowed every REFUSAL the resolver raises on
569
+ * purpose (`guardedPhysicalIdFallback`'s ARN / URL shape hard-fail, the
570
+ * `--strict-getatt` rejection, `rejectPlaceholderArnAttribute`), so a template
571
+ * that hard-fails when the reference sits in a resource property silently
572
+ * degraded to shipping a literal `${Resource.Attribute}` to AWS when the
573
+ * IDENTICAL reference was written inside an `Fn::Sub`.
574
+ *
575
+ * Throwing this class rather than a bare `Error` is what lets that catch
576
+ * re-raise a refusal (carrying its own message and remedy) while leaving the
577
+ * not-found path on warn-and-keep. Nothing else branches on it.
578
+ */
579
+ var IntrinsicResolutionRefusalError = class IntrinsicResolutionRefusalError extends CdkdError {
580
+ constructor(message, cause) {
581
+ super(message, "INTRINSIC_RESOLUTION_REFUSAL", cause);
582
+ this.name = "IntrinsicResolutionRefusalError";
583
+ Object.setPrototypeOf(this, IntrinsicResolutionRefusalError.prototype);
584
+ }
585
+ };
586
+ /**
587
+ * Dependency resolution errors
588
+ */
589
+ var DependencyError = class DependencyError extends CdkdError {
590
+ constructor(message, cause) {
591
+ super(message, "DEPENDENCY_ERROR", cause);
592
+ this.name = "DependencyError";
593
+ Object.setPrototypeOf(this, DependencyError.prototype);
594
+ }
595
+ };
596
+ /**
597
+ * Configuration errors
598
+ */
599
+ var ConfigError = class ConfigError extends CdkdError {
600
+ constructor(message, cause) {
601
+ super(message, "CONFIG_ERROR", cause);
602
+ this.name = "ConfigError";
603
+ Object.setPrototypeOf(this, ConfigError.prototype);
604
+ }
605
+ };
606
+ /**
607
+ * Signals a partial-failure outcome that should map to exit code 2 (not 1).
608
+ *
609
+ * Used by `cdkd destroy` and `cdkd state destroy` when one or more
610
+ * per-resource deletes failed but the overall command finished its work
611
+ * (state.json is preserved, the rest of the stack was deleted, and the
612
+ * user can re-run to clean up the remaining resources).
272
613
  *
273
614
  * Exit code conventions:
274
615
  * - 0: command completed successfully, no resources left in error state.
@@ -310,6 +651,27 @@ var PartialFailureError = class PartialFailureError extends CdkdError {
310
651
  * success rather than fatal — the rest of the drifted resources still
311
652
  * had their `update` invoked, and the user has a clear next step printed
312
653
  * for the unsupported one.
654
+ *
655
+ * **Marked non-retryable in the constructor** (issue
656
+ * [#1838](https://github.com/go-to-k/cdkd/issues/1838)). This is a
657
+ * deterministic, cdkd-authored refusal — the deploy engine matches it BY
658
+ * CLASS to choose the `--replace` DELETE+CREATE fallback, and `cdkd drift
659
+ * --revert` reports it as its own per-resource outcome — so it can never
660
+ * succeed on a retry, whatever its message happens to say. That last clause
661
+ * is the whole point: the message interpolates the resource's LOGICAL ID,
662
+ * and the retry classifiers match by SUBSTRING, so a perfectly ordinary
663
+ * composite CDK id like `MyDependencyViolationSub` made
664
+ * `isRetryableTransientError` return true (`DependencyViolation` is the only
665
+ * whitespace-free entry in `RETRYABLE_ERROR_MESSAGE_PATTERNS`) and burned the
666
+ * full generic schedule — 8 retries, ~47s of pure sleep — on a path that was
667
+ * never going to succeed, before the fallback the error exists to trigger was
668
+ * even reached.
669
+ *
670
+ * Marking in the CONSTRUCTOR rather than at each `throw` is deliberate: ~20
671
+ * providers raise this from inside the `update()` call `deploy-engine.ts`
672
+ * wraps in `withRetry` (plus `drift.ts`'s `--revert` update), and a per-site
673
+ * marker is one forgotten call away from re-opening the hole for exactly one
674
+ * provider.
313
675
  */
314
676
  var ResourceUpdateNotSupportedError = class ResourceUpdateNotSupportedError extends CdkdError {
315
677
  exitCode = 2;
@@ -330,6 +692,7 @@ var ResourceUpdateNotSupportedError = class ResourceUpdateNotSupportedError exte
330
692
  this.suggestion = suggestion;
331
693
  this.name = "ResourceUpdateNotSupportedError";
332
694
  Object.setPrototypeOf(this, ResourceUpdateNotSupportedError.prototype);
695
+ markNonRetryable(this);
333
696
  }
334
697
  };
335
698
  /**
@@ -8115,346 +8478,6 @@ var ReplacementRulesRegistry = class {
8115
8478
  }
8116
8479
  };
8117
8480
 
8118
- //#endregion
8119
- //#region src/deployment/retryable-errors.ts
8120
- /**
8121
- * The **IAM-propagation** subset of {@link RETRYABLE_ERROR_MESSAGE_PATTERNS}:
8122
- * an AWS service rejecting a call because a just-created IAM entity (role,
8123
- * trust policy, inline policy, instance profile, principal) has not propagated
8124
- * to that service's authorization layer yet.
8125
- *
8126
- * Kept as its own array — and composed back into the full transient table
8127
- * below — so there is exactly ONE list per pattern (no parallel classifier to
8128
- * drift). It exists because this class has a materially different RECOVERY
8129
- * SHAPE from the other transient errors: it resolves in single-digit seconds,
8130
- * so `withRetry` polls it on a dense sub-second schedule instead of the
8131
- * generic 1s/2s/4s/8s exponential backoff (which is right for throttling and
8132
- * for long resource-state transitions, and wrong here — see
8133
- * {@link file://../deployment/retry.ts}).
8134
- *
8135
- * When adding a new pattern: put it here if the fix is "wait a moment and ask
8136
- * IAM again", and in `OTHER_TRANSIENT_ERROR_MESSAGE_PATTERNS` otherwise. A
8137
- * misfiled entry only changes the retry CADENCE, never whether the error is
8138
- * retryable at all.
8139
- */
8140
- const IAM_PROPAGATION_ERROR_MESSAGE_PATTERNS = [
8141
- "cannot be assumed",
8142
- "Firehose is unable to assume role",
8143
- "is unable to assume provided role",
8144
- "is unable to assume the role",
8145
- "security token included in the request is invalid. (Service:",
8146
- "role defined for the function",
8147
- "not authorized to perform",
8148
- "execution role",
8149
- "trust policy",
8150
- "Role validation failed",
8151
- "does not have required permissions",
8152
- "Trusted Entity",
8153
- "Invalid principal in policy",
8154
- "Verify in IAM that the role has adequate trust relationships",
8155
- "The user with name",
8156
- "Policy Error: PrincipalNotFound",
8157
- "Invalid value for the parameter Policy",
8158
- "required permissions for: ENHANCED_MONITORING",
8159
- "Caught ServiceAccessDeniedException",
8160
- "permissions required to assume the role",
8161
- "authorized to assume the provided role",
8162
- "Cannot access stream",
8163
- "Please ensure the role can perform",
8164
- "KMS key is invalid for CreateGrant",
8165
- "Policy contains a statement with one or more invalid principals",
8166
- "Invalid IAM Instance Profile",
8167
- "Invalid InstanceProfile",
8168
- "Failed to authorize instance profile",
8169
- "is not a valid role to allow SNS"
8170
- ];
8171
- /**
8172
- * The NON-IAM-propagation half of {@link RETRYABLE_ERROR_MESSAGE_PATTERNS}:
8173
- * transient failures whose recovery window is either long (SQS's 60s same-name
8174
- * cooldown, a resource still leaving a Pending/Creating state) or genuinely
8175
- * load-related (throttling), where hammering AWS with dense retries is harmful
8176
- * and exponential backoff is the correct shape.
8177
- */
8178
- const OTHER_TRANSIENT_ERROR_MESSAGE_PATTERNS = [
8179
- "currently in the following state: Pending",
8180
- "has dependencies and cannot be deleted",
8181
- "can't be deleted since it has",
8182
- "DependencyViolation",
8183
- "does not exist",
8184
- "Schema is currently being altered",
8185
- "conflicting conditional operation",
8186
- "scheduled for deletion",
8187
- "Could not deliver test message",
8188
- "wait 60 seconds",
8189
- "concurrent update operation",
8190
- "because it is in use",
8191
- "Rate exceeded",
8192
- "is marked disabled for mutation",
8193
- "There is an operation running on the Cluster",
8194
- "Unable to complete operation due to concurrent modification"
8195
- ];
8196
- /**
8197
- * Patterns that mark an AWS error as a transient/retryable failure.
8198
- * Each entry is a substring match against the error message; all of these
8199
- * are situations where the same call typically succeeds after a short delay
8200
- * because of eventual consistency or just-created-dependency propagation.
8201
- *
8202
- * Composed from the two halves above so retryability has ONE source of truth
8203
- * while `withRetry` can still pick a per-class backoff cadence.
8204
- */
8205
- const RETRYABLE_ERROR_MESSAGE_PATTERNS = [...IAM_PROPAGATION_ERROR_MESSAGE_PATTERNS, ...OTHER_TRANSIENT_ERROR_MESSAGE_PATTERNS];
8206
- /**
8207
- * HTTP status codes that always indicate a transient failure worth retrying.
8208
- * 429 = Too Many Requests (throttle), 503 = Service Unavailable.
8209
- */
8210
- const RETRYABLE_HTTP_STATUS_CODES = /* @__PURE__ */ new Set([429, 503]);
8211
- /**
8212
- * AWS SDK v3 canonical throttling error names. Mirrors
8213
- * `@aws-sdk/service-error-classification`'s `THROTTLING_ERROR_CODES` — any
8214
- * error (or wrapped cause) whose `name` is one of these is a transient rate-
8215
- * limit rejection worth retrying with backoff. Detecting by NAME is more
8216
- * robust than by HTTP status because most AWS throttles surface as HTTP 400
8217
- * (not 429) with the throttling signal carried only in the error code / name
8218
- * (e.g. SSM `ThrottlingException` for the `Rate exceeded` message).
8219
- */
8220
- const THROTTLING_ERROR_NAMES = /* @__PURE__ */ new Set([
8221
- "BandwidthLimitExceeded",
8222
- "EC2ThrottledException",
8223
- "LimitExceededException",
8224
- "PriorRequestNotComplete",
8225
- "ProvisionedThroughputExceededException",
8226
- "RequestLimitExceeded",
8227
- "RequestThrottled",
8228
- "RequestThrottledException",
8229
- "SlowDown",
8230
- "ThrottledException",
8231
- "Throttling",
8232
- "ThrottlingException",
8233
- "TooManyRequestsException",
8234
- "TransactionInProgressException"
8235
- ]);
8236
- /**
8237
- * Marker for an error cdkd raised as a DELIBERATE refusal rather than as a
8238
- * relayed AWS failure (issue [#1778](https://github.com/go-to-k/cdkd/issues/1778)).
8239
- *
8240
- * Every classifier below is SUBSTRING-based, which is the right shape for
8241
- * relaying a vendor's message and the wrong one for cdkd's own prose: a
8242
- * refusal message is assembled from values cdkd does not control — a provider
8243
- * `reason`, a state-borne physicalId, a template logical id — and any of them
8244
- * can happen to contain a retryable pattern. Measured: a resource named
8245
- * `MyDependencyViolationSub` puts `DependencyViolation` in the message, so a
8246
- * deterministic refusal was classified transient and burned the whole backoff
8247
- * schedule before failing exactly as it would have immediately. Keeping the
8248
- * offending values OUT of the message narrows that surface but cannot close
8249
- * it, because a message with no identifiers at all is not diagnosable.
8250
- *
8251
- * A marker inverts the burden: the raiser STATES that the error is terminal,
8252
- * so no wording can make it retryable. Deliberately a `Symbol.for` key —
8253
- * global-registry symbols survive a duplicated module instance (dual
8254
- * bundling), where a module-local symbol would silently stop matching — and
8255
- * non-enumerable, so it cannot leak into a serialized error payload.
8256
- */
8257
- const NON_RETRYABLE_MARKER = Symbol.for("cdkd.nonRetryable");
8258
- /**
8259
- * Mark a cdkd-authored refusal as terminal and return it, for
8260
- * `throw markNonRetryable(new ProvisioningError(...))`.
8261
- *
8262
- * Reach for it when the error means "this cannot succeed on a retry" as a
8263
- * matter of cdkd's own logic — NOT for a relayed AWS failure, whose
8264
- * retryability is the classifiers' business.
8265
- *
8266
- * KNOWN LIVE INSTANCE NOT YET COVERED: `ResourceUpdateNotSupportedError`
8267
- * (`src/utils/error-handler.ts`) interpolates the logical id and is thrown by
8268
- * ~20 providers from inside the retried `update()` in `deploy-engine.ts`, so a
8269
- * stack with a resource named e.g. `MyDependencyViolationSub` burns the full
8270
- * ~47s backoff schedule before the `--replace` fallback is even reached. That
8271
- * is a LIVE occurrence of the class this marker exists for, unlike the
8272
- * latent-today SNS abort that motivated it. Marking it belongs in that error's
8273
- * constructor, in a file this change does not own; tracked separately.
8274
- */
8275
- function markNonRetryable(error) {
8276
- if (!Object.isExtensible(error)) return error;
8277
- Object.defineProperty(error, NON_RETRYABLE_MARKER, {
8278
- value: true,
8279
- enumerable: false,
8280
- configurable: true,
8281
- writable: false
8282
- });
8283
- return error;
8284
- }
8285
- /**
8286
- * True when the error, or anything in its bounded `.cause` chain, was marked
8287
- * by {@link markNonRetryable}.
8288
- *
8289
- * The chain walk mirrors {@link isThrottlingError}'s: cdkd wraps errors, so a
8290
- * marked refusal can end up one or more links deep, and a marker that stopped
8291
- * counting after a single wrap would be a fence that quietly falls open.
8292
- *
8293
- * That reach is DIRECTIONAL, and the upward direction is a hazard worth
8294
- * stating. Downward — a marked refusal wrapped by an outer error — is the
8295
- * intended case and stays terminal. UPWARD is the inverse: wrapping a marked
8296
- * refusal as the `cause` of a genuinely RETRYABLE outer error
8297
- * (`new Error(msg, { cause: markedRefusal })`) makes the outer error terminal
8298
- * too, because this walk finds the marker on the cause. Unconstructible today
8299
- * (`ProvisioningError` is built with no `cause` at the one marking site), and
8300
- * the failure is fail-fast rather than silent, but a future wrapper that
8301
- * carries a marked cause into a transient error would stop retrying something
8302
- * that should retry. Strip or re-raise the cause there rather than nesting it.
8303
- */
8304
- function isMarkedNonRetryable(error) {
8305
- let current = error;
8306
- for (let depth = 0; depth < 5 && current != null; depth++) {
8307
- if (typeof current === "object" || typeof current === "function") {
8308
- if (current[NON_RETRYABLE_MARKER] === true) return true;
8309
- }
8310
- current = current.cause;
8311
- }
8312
- return false;
8313
- }
8314
- /**
8315
- * Walk the error + its `.cause` chain (bounded) looking for a rate-limit
8316
- * signal — either an AWS SDK v3 throttling error `name`
8317
- * ({@link THROTTLING_ERROR_NAMES}) or a retryable HTTP status
8318
- * ({@link RETRYABLE_HTTP_STATUS_CODES}) on `$metadata`.
8319
- *
8320
- * cdkd wraps the original AWS error in a `ProvisioningError`, so the signal is
8321
- * typically one cause-link deep; the bounded walk also tolerates SDK errors
8322
- * that nest a `$response`/cause without exploding on a cyclic chain.
8323
- *
8324
- * BOTH signals are checked at EVERY depth. An earlier version checked the name
8325
- * to depth 5 but the HTTP status only at depths 0 and 1, so a 429 nested two
8326
- * links deep was missed.
8327
- */
8328
- function isThrottlingError(error) {
8329
- let current = error;
8330
- for (let depth = 0; depth < 5 && current != null; depth++) {
8331
- const name = current.name;
8332
- if (typeof name === "string" && THROTTLING_ERROR_NAMES.has(name)) return true;
8333
- const status = current.$metadata?.httpStatusCode;
8334
- if (status !== void 0 && RETRYABLE_HTTP_STATUS_CODES.has(status)) return true;
8335
- current = current.cause;
8336
- }
8337
- return false;
8338
- }
8339
- /**
8340
- * Determine whether an AWS error should be retried.
8341
- *
8342
- * Checks (in order):
8343
- * 0. {@link isMarkedNonRetryable} — a cdkd-authored refusal is terminal by
8344
- * declaration, ahead of every message / name heuristic below. FIRST on
8345
- * purpose: the marker states the error cannot succeed on a retry, so
8346
- * nothing a later check reads out of the message can overturn it.
8347
- * 1. Rate-limit signal on the error or any wrapped cause — throttling error
8348
- * `name` or retryable HTTP status (most AWS throttles are HTTP 400, not
8349
- * 429, so the name check carries most of the weight). See
8350
- * {@link isThrottlingError}.
8351
- * 2. Substring match against {@link RETRYABLE_ERROR_MESSAGE_PATTERNS}
8352
- */
8353
- function isRetryableTransientError(error, message) {
8354
- if (isMarkedNonRetryable(error)) return false;
8355
- if (isThrottlingError(error)) return true;
8356
- return RETRYABLE_ERROR_MESSAGE_PATTERNS.some((p) => message.includes(p));
8357
- }
8358
- /**
8359
- * True when the message is a just-created-IAM-entity propagation rejection
8360
- * ({@link IAM_PROPAGATION_ERROR_MESSAGE_PATTERNS}).
8361
- *
8362
- * This does NOT decide retryability — every pattern it matches is already in
8363
- * {@link RETRYABLE_ERROR_MESSAGE_PATTERNS}. It only selects the retry CADENCE:
8364
- * `withRetry` polls this class densely (sub-second initial delay, low cap)
8365
- * because IAM propagation resolves in single-digit seconds, whereas the
8366
- * generic exponential schedule is tuned for throttling and long resource-state
8367
- * transitions.
8368
- *
8369
- * Deliberately message-only (no error-object inspection): the propagation
8370
- * signal is always carried in the vendor's message text, and cdkd wraps the
8371
- * original error in a `ProvisioningError` that preserves it.
8372
- */
8373
- function isIamPropagationError(message) {
8374
- return IAM_PROPAGATION_ERROR_MESSAGE_PATTERNS.some((p) => message.includes(p));
8375
- }
8376
- /**
8377
- * Match the "already exists" name-collision signature raised when a create
8378
- * targets a physical name still held by another resource (or by the same
8379
- * name's not-yet-released tombstone after an async delete).
8380
- *
8381
- * Deliberately NOT part of {@link RETRYABLE_ERROR_MESSAGE_PATTERNS}: a name
8382
- * collision is only worth retrying at the specific re-create sites that just
8383
- * deleted the old holder (the deploy engine's --replace delete-first fallback
8384
- * and the rollback executor's reverse-replacement) — everywhere else it is a
8385
- * genuine conflict that must fail fast. Shared by those sites' collision
8386
- * detection + retry filters so a signature extension lands in one place.
8387
- *
8388
- * The optional `s` is load-bearing, not defensive spelling (issue #1625):
8389
- * Lambda's `CreateFunction` raises `ResourceConflictException: Function
8390
- * already exist: <name>` — SINGULAR — so the `already exists` form missed it
8391
- * entirely and NO Lambda function could take the collision path. The
8392
- * consequence was not a cosmetic message: a property-driven replacement of a
8393
- * Lambda (dropping `DurableConfig`, changing `TenancyConfig`) create-firsts
8394
- * into its own still-live name, the raw `ResourceConflictException` escaped
8395
- * instead of the actionable `NAMED_REPLACEMENT_COLLISION` error, and
8396
- * `cdkd deploy --replace`'s delete-first fallback never fired — so the
8397
- * replacement was unperformable by any flag. Verified against real AWS
8398
- * (us-east-1, 2026-08-12) by creating one function name twice.
8399
- *
8400
- * Two fences keep the widened arm from crediting a NON-collision, which
8401
- * matters because the sites that consult it react DESTRUCTIVELY (the
8402
- * `--replace` delete-first fallback deletes the old resource):
8403
- * - `\b` after `exists?` rejects a participle ("already existed as a draft");
8404
- * - the lookbehind rejects a NEGATED or MODAL phrase — "the bucket does NOT
8405
- * already exist", "the destination bucket MUST already exist" — which the
8406
- * bare pattern matched. The modal form is the one that bites: a create
8407
- * rejected for a missing PREREQUISITE would be reported to the user as a
8408
- * name collision pointing at `--replace`, and following that advice
8409
- * deletes the live old resource before the re-create fails again for the
8410
- * same reason, leaving it absent from AWS with state still recording it.
8411
- * The error-CODE arm stays exact (`AlreadyExists`): the singular
8412
- * `AlreadyExist` is not an AWS code spelling, and loosening it would match
8413
- * inside unrelated identifiers.
8414
- *
8415
- * Classification stays MESSAGE-based rather than moving to the exception
8416
- * NAME, and that is load-bearing here rather than inherited: Lambda raises
8417
- * `ResourceConflictException` for a function in a PENDING state too (see
8418
- * `lambda-function-provider.ts`), so keying on the name would classify a
8419
- * transient state conflict as a collision and delete a live function under
8420
- * `--replace`.
8421
- */
8422
- function isNameCollisionError(message) {
8423
- return /(?<!\b(?:must|not|should|may|cannot)\s)already exists?\b/i.test(message) || message.includes("AlreadyExists");
8424
- }
8425
- /**
8426
- * Match the SQS same-name re-creation cooldown: after `DeleteQueue`, creating
8427
- * a queue with the SAME name inside ~60s fails with
8428
- * `AWS.SimpleQueueService.QueueDeletedRecently` ("You must wait 60 seconds
8429
- * after deleting a queue before you can create another with the same name").
8430
- *
8431
- * The generic transient table above already carries 'wait 60 seconds' for
8432
- * plain CREATEs (rapid destroy → redeploy loops), but the delete-then-re-create
8433
- * sites (the deploy engine's --replace delete-first fallback and the rollback
8434
- * executor's reverse-replacement) override the retry filter with
8435
- * {@link isNameCollisionError}, which this signature does NOT match — so a
8436
- * replacement revert used to fail fast mid-flight with the resource absent
8437
- * from both AWS and state (issue #1206). Those sites now OR this matcher into
8438
- * their retry filter, with a schedule long enough to cover the 60s window.
8439
- *
8440
- * Kept separate from {@link isNameCollisionError} on purpose: a cooldown at a
8441
- * create-first site must NOT be treated as a collision (deleting the new
8442
- * resource would not release the cooldown on the old name).
8443
- */
8444
- function isNameCooldownError(message) {
8445
- return message.includes("QueueDeletedRecently") || message.includes("wait 60 seconds");
8446
- }
8447
- /**
8448
- * Retry filter for the delete-then-re-create sites: the old name holder was
8449
- * just deleted, so both the late name release ("already exists" from an async
8450
- * delete) and the SQS 60s name cooldown are worth waiting out. Pair with a
8451
- * schedule that covers the full cooldown window (maxRetries 8, delays
8452
- * 2s/4s/8s then capped at 10s ≈ 64s total sleep).
8453
- */
8454
- function isRecreateRetryableError(message) {
8455
- return isNameCollisionError(message) || isNameCooldownError(message);
8456
- }
8457
-
8458
8481
  //#endregion
8459
8482
  //#region src/deployment/retry.ts
8460
8483
  /**
@@ -13971,7 +13994,7 @@ var CloudControlProvider = class {
13971
13994
  if (context?.finalSnapshotIdentifier !== void 0) throw new ProvisioningError(`${logicalId} (${resourceType}) requires a final snapshot (DeletionPolicy: Snapshot), but the Cloud Control API delete route has no final-snapshot parameter. Re-run with --skip-final-snapshot after snapshotting manually, or retain the resource.`, resourceType, logicalId, physicalId);
13972
13995
  if (context?.removeProtection === true && resourceType === "AWS::AutoScaling::AutoScalingGroup") {
13973
13996
  this.logger.debug(`Delegating protected AutoScalingGroup ${logicalId} delete to the SDK ASGProvider (Cloud Control cannot force-delete a protected ASG)`);
13974
- const { ASGProvider } = await import("./asg-provider-Tu4sDbhu.js").then((n) => n.n);
13997
+ const { ASGProvider } = await import("./asg-provider-D5ZBSgh2.js").then((n) => n.n);
13975
13998
  return await new ASGProvider().delete(logicalId, physicalId, resourceType, _properties, context);
13976
13999
  }
13977
14000
  const isProtectedEc2Instance = context?.removeProtection === true && resourceType === "AWS::EC2::Instance";
@@ -21025,7 +21048,7 @@ const FLUSH_INTERVAL_MS = 2e3;
21025
21048
  const FLUSH_EVENT_THRESHOLD = 50;
21026
21049
  /** Build-time cdkd version, with a dev fallback for non-built contexts. */
21027
21050
  function getCdkdVersion() {
21028
- return "0.283.8";
21051
+ return "0.283.9";
21029
21052
  }
21030
21053
  /**
21031
21054
  * Generate a time-sortable unique run id, e.g.
@@ -23241,5 +23264,5 @@ var DeployEngine = class {
23241
23264
  };
23242
23265
 
23243
23266
  //#endregion
23244
- export { coerceCfnBoolean as $, resolveAutoAssetStorage as $t, cyan as A, LocalMigrateError as An, createAssetRedirectResolver as At, findSilentDropProperties as B, StackTerminationProtectionError as Bn, buildDenyExternalAccessPolicy as Bt, makeCanonicalizePropertiesFn as C, setAwsClients as Cn, S3StateBackend as Ct, renderStatefulReason as D, DependencyError as Dn, stringifyValue as Dt, isStatefulRecreateTargetSync as E, ConfigError as En, AssetPublisher as Et, IAMRoleProvider as F, PartialFailureError as Fn, ensureAssetStorage as Ft, IntrinsicFunctionResolver as G, normalizeAwsError as Gn, runDockerStreaming as Gt, slowCcOperationTimeoutMs as H, SynthesisError as Hn, formatDockerLoginError as Ht, collectInlinePolicyNamesManagedBySiblings as I, ProvisioningError as In, getBootstrapMarkerKey as It, refStateLookupFromResource as J, Synthesizer as Jt, cfnRefValueFromPhysicalId as K, withErrorHandling as Kn, AssetManifestLoader as Kt, clearOnUpdateRemoval as L, ResourceTimeoutError as Ln, parseBootstrapMarker as Lt, green as M, LockError as Mn, rewriteTemplateAssetReferences as Mt, red as N, MissingCdkCliError as Nn, AssetModeResolver as Nt, formatResourceLine as O, DeployCancelledError as On, WorkGraph as Ot, yellow as P, NestedStackChildDirectDestroyError as Pn, BOOTSTRAP_MARKER_PREFIX as Pt, assertRegionMatch as Q, resolveApp as Qt, ProviderRegistry as R, ResourceUpdateNotSupportedError as Rn, validateAssetBucketName as Rt, unsupportedFinalSnapshotError as S, resetAwsClients as Sn, LockManager as St, MULTI_REGION_RECREATE_BLOCKED_TYPES as T, CdkdError as Tn, shouldRetainResource as Tt, disableInstanceApiTermination as U, formatError as Un, getDockerCmd as Ut, CloudControlProvider as V, StateError as Vn, buildDockerImage as Vt, isTerminationProtectionPropagationError as W, isCdkdError as Wn, runDockerForeground as Wt, normalizeAwsTagsToCfn as X, getDefaultStateBucketName as Xt, WAFv2WebACLProvider as Y, synthesisStatusMessage as Yt, resolveExplicitPhysicalId as Z, getLegacyStateBucketName as Zt, buildFinalSnapshotIdentifier as _, processStackMessages as _n, isRetryableTransientError as _t, DeploymentEventsStore as a, stateBucketExistenceConfirmed as an, requireConfigObject as at, isFinalSnapshotError as b, AwsClients as bn, DagBuilder as bt, replayFailedOperations as c, CFN_TEMPLATE_URL_LIMIT as cn, s3BucketDomainName as ct, deleteSkipReason as d, uploadCfnTemplate as dn, s3BucketWebsiteUrl as dt, resolveCaptureObservedState as en, configBooleanRefusal as et, withResourceDeadline as f, expectedOwnerParam as fn, applyRoleArnIfSet as ft, PRE_DELETE_SNAPSHOT_TYPES as g, AssemblyReader as gn, isMarkedNonRetryable as gt, ATOMIC_FINAL_SNAPSHOT_TYPES as h, derivePartitionAndUrlSuffix as hn, withRetry as ht, DeploymentEventsReader as i, resolveUseCdkBootstrapAssets as in, requireConfigArray as it, gray as j, LocalStartServiceError as jn, loadPublishableAssetManifest as jt, bold as k, LocalInvokeBuildError as kn, buildAssetRedirectMap as kt, replayRollback as l, MIGRATE_TMP_PREFIX as ln, s3BucketDualStackDomainName as lt, computeImplicitDeleteEdges as m, canonicalizeRegion as mn, describeTypeWithThrottleRetry as mt, DEFAULT_RESOURCE_WARN_AFTER_MS as n, resolveStateBucketWithDefault as nn, readConfigString as nt, planFailedOps as o, warnDeprecatedNoPrefixCliFlag as on, requireConfigString as ot, IMPLICIT_DELETE_DEPENDENCIES as p, PARTITION_TABLE as pn, DiffCalculator as pt, getAccountInfo as q, __exportAll as qn, getDockerImageBySourceHash as qt, DeployEngine as r, resolveStateBucketWithDefaultAndSource as rn, replayWarn as rt, planRollback as s, CFN_TEMPLATE_BODY_LIMIT as sn, s3BucketArn as st, DEFAULT_RESOURCE_TIMEOUT_MS as t, resolveSkipPrefix as tn, configStringRefusal as tt, UNSPECIFIED_SKIP_REASON as u, findLargeInlineResources as un, s3BucketRegionalDomainName as ut, ccRoutedFinalSnapshotError as v, clearBucketRegionCache as vn, isThrottlingError as vt, extractDeploymentEventError as w, AssetError as wn, rebuildClientForBucketRegion as wt, refusesFinalSnapshot as x, getAwsClients as xn, TemplateParser as xt, createPreDeleteFinalSnapshot as y, resolveBucketRegion as yn, markNonRetryable as yt, findActionableSilentDrops as z, StackHasActiveImportsError as zn, validateContainerRepoName as zt };
23245
- //# sourceMappingURL=deploy-engine-D95wmKN8.js.map
23267
+ export { coerceCfnBoolean as $, resolveStateBucketWithDefaultAndSource as $t, cyan as A, NestedStackChildDirectDestroyError as An, BOOTSTRAP_MARKER_PREFIX as At, findSilentDropProperties as B, isCdkdError as Bn, runDockerForeground as Bt, makeCanonicalizePropertiesFn as C, DependencyError as Cn, stringifyValue as Ct, renderStatefulReason as D, LocalStartServiceError as Dn, loadPublishableAssetManifest as Dt, isStatefulRecreateTargetSync as E, LocalMigrateError as En, createAssetRedirectResolver as Et, IAMRoleProvider as F, StackHasActiveImportsError as Fn, validateContainerRepoName as Ft, IntrinsicFunctionResolver as G, isThrottlingError as Gn, synthesisStatusMessage as Gt, slowCcOperationTimeoutMs as H, withErrorHandling as Hn, AssetManifestLoader as Ht, collectInlinePolicyNamesManagedBySiblings as I, StackTerminationProtectionError as In, buildDenyExternalAccessPolicy as It, refStateLookupFromResource as J, resolveApp as Jt, cfnRefValueFromPhysicalId as K, markNonRetryable as Kn, getDefaultStateBucketName as Kt, clearOnUpdateRemoval as L, StateError as Ln, buildDockerImage as Lt, green as M, ProvisioningError as Mn, getBootstrapMarkerKey as Mt, red as N, ResourceTimeoutError as Nn, parseBootstrapMarker as Nt, formatResourceLine as O, LockError as On, rewriteTemplateAssetReferences as Ot, yellow as P, ResourceUpdateNotSupportedError as Pn, validateAssetBucketName as Pt, assertRegionMatch as Q, resolveStateBucketWithDefault as Qt, ProviderRegistry as R, SynthesisError as Rn, formatDockerLoginError as Rt, unsupportedFinalSnapshotError as S, ConfigError as Sn, AssetPublisher as St, MULTI_REGION_RECREATE_BLOCKED_TYPES as T, LocalInvokeBuildError as Tn, buildAssetRedirectMap as Tt, disableInstanceApiTermination as U, isMarkedNonRetryable as Un, getDockerImageBySourceHash as Ut, CloudControlProvider as V, normalizeAwsError as Vn, runDockerStreaming as Vt, isTerminationProtectionPropagationError as W, isRetryableTransientError as Wn, Synthesizer as Wt, normalizeAwsTagsToCfn as X, resolveCaptureObservedState as Xt, WAFv2WebACLProvider as Y, resolveAutoAssetStorage as Yt, resolveExplicitPhysicalId as Z, resolveSkipPrefix as Zt, buildFinalSnapshotIdentifier as _, getAwsClients as _n, TemplateParser as _t, DeploymentEventsStore as a, MIGRATE_TMP_PREFIX as an, requireConfigObject as at, isFinalSnapshotError as b, AssetError as bn, rebuildClientForBucketRegion as bt, replayFailedOperations as c, expectedOwnerParam as cn, s3BucketDomainName as ct, deleteSkipReason as d, derivePartitionAndUrlSuffix as dn, s3BucketWebsiteUrl as dt, resolveUseCdkBootstrapAssets as en, configBooleanRefusal as et, withResourceDeadline as f, AssemblyReader as fn, applyRoleArnIfSet as ft, PRE_DELETE_SNAPSHOT_TYPES as g, AwsClients as gn, DagBuilder as gt, ATOMIC_FINAL_SNAPSHOT_TYPES as h, resolveBucketRegion as hn, withRetry as ht, DeploymentEventsReader as i, CFN_TEMPLATE_URL_LIMIT as in, requireConfigArray as it, gray as j, PartialFailureError as jn, ensureAssetStorage as jt, bold as k, MissingCdkCliError as kn, AssetModeResolver as kt, replayRollback as l, PARTITION_TABLE as ln, s3BucketDualStackDomainName as lt, computeImplicitDeleteEdges as m, clearBucketRegionCache as mn, describeTypeWithThrottleRetry as mt, DEFAULT_RESOURCE_WARN_AFTER_MS as n, warnDeprecatedNoPrefixCliFlag as nn, readConfigString as nt, planFailedOps as o, findLargeInlineResources as on, requireConfigString as ot, IMPLICIT_DELETE_DEPENDENCIES as p, processStackMessages as pn, DiffCalculator as pt, getAccountInfo as q, __exportAll as qn, getLegacyStateBucketName as qt, DeployEngine as r, CFN_TEMPLATE_BODY_LIMIT as rn, replayWarn as rt, planRollback as s, uploadCfnTemplate as sn, s3BucketArn as st, DEFAULT_RESOURCE_TIMEOUT_MS as t, stateBucketExistenceConfirmed as tn, configStringRefusal as tt, UNSPECIFIED_SKIP_REASON as u, canonicalizeRegion as un, s3BucketRegionalDomainName as ut, ccRoutedFinalSnapshotError as v, resetAwsClients as vn, LockManager as vt, extractDeploymentEventError as w, DeployCancelledError as wn, WorkGraph as wt, refusesFinalSnapshot as x, CdkdError as xn, shouldRetainResource as xt, createPreDeleteFinalSnapshot as y, setAwsClients as yn, S3StateBackend as yt, findActionableSilentDrops as z, formatError as zn, getDockerCmd as zt };
23268
+ //# sourceMappingURL=deploy-engine-CWt7h7Nc.js.map