alchemy 2.0.0-beta.64 → 2.0.0-beta.65

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (165) hide show
  1. package/lib/AWS/EC2/InternetGateway.d.ts.map +1 -1
  2. package/lib/AWS/EC2/InternetGateway.js +8 -3
  3. package/lib/AWS/EC2/InternetGateway.js.map +1 -1
  4. package/lib/AWS/EC2/LingeringEnis.d.ts +26 -0
  5. package/lib/AWS/EC2/LingeringEnis.d.ts.map +1 -0
  6. package/lib/AWS/EC2/LingeringEnis.js +128 -0
  7. package/lib/AWS/EC2/LingeringEnis.js.map +1 -0
  8. package/lib/AWS/EC2/SecurityGroup.d.ts.map +1 -1
  9. package/lib/AWS/EC2/SecurityGroup.js +14 -14
  10. package/lib/AWS/EC2/SecurityGroup.js.map +1 -1
  11. package/lib/AWS/EC2/Subnet.d.ts.map +1 -1
  12. package/lib/AWS/EC2/Subnet.js +13 -21
  13. package/lib/AWS/EC2/Subnet.js.map +1 -1
  14. package/lib/AWS/ECR/Image.d.ts +1 -0
  15. package/lib/AWS/ECR/Image.d.ts.map +1 -1
  16. package/lib/AWS/EKS/Addon.d.ts.map +1 -1
  17. package/lib/AWS/EKS/Addon.js +15 -3
  18. package/lib/AWS/EKS/Addon.js.map +1 -1
  19. package/lib/AWS/EKS/Deployment.d.ts +10 -1
  20. package/lib/AWS/EKS/Deployment.d.ts.map +1 -1
  21. package/lib/AWS/EKS/Deployment.js +10 -4
  22. package/lib/AWS/EKS/Deployment.js.map +1 -1
  23. package/lib/AWS/EKS/HyperPod.d.ts +57 -0
  24. package/lib/AWS/EKS/HyperPod.d.ts.map +1 -0
  25. package/lib/AWS/EKS/HyperPod.js +46 -0
  26. package/lib/AWS/EKS/HyperPod.js.map +1 -0
  27. package/lib/AWS/EKS/Job.d.ts +10 -1
  28. package/lib/AWS/EKS/Job.d.ts.map +1 -1
  29. package/lib/AWS/EKS/Job.js +16 -6
  30. package/lib/AWS/EKS/Job.js.map +1 -1
  31. package/lib/AWS/EKS/internal/client.d.ts +4 -3
  32. package/lib/AWS/EKS/internal/client.d.ts.map +1 -1
  33. package/lib/AWS/EKS/internal/client.js +33 -2
  34. package/lib/AWS/EKS/internal/client.js.map +1 -1
  35. package/lib/AWS/Lambda/Function.d.ts.map +1 -1
  36. package/lib/AWS/Lambda/Function.js +1 -1
  37. package/lib/AWS/Lambda/Function.js.map +1 -1
  38. package/lib/AWS/Providers.d.ts.map +1 -1
  39. package/lib/AWS/Providers.js +4 -1
  40. package/lib/AWS/Providers.js.map +1 -1
  41. package/lib/AWS/SageMaker/Cluster.d.ts +327 -0
  42. package/lib/AWS/SageMaker/Cluster.d.ts.map +1 -0
  43. package/lib/AWS/SageMaker/Cluster.js +469 -0
  44. package/lib/AWS/SageMaker/Cluster.js.map +1 -0
  45. package/lib/AWS/SageMaker/ClusterSchedulerConfig.d.ts +96 -0
  46. package/lib/AWS/SageMaker/ClusterSchedulerConfig.d.ts.map +1 -0
  47. package/lib/AWS/SageMaker/ClusterSchedulerConfig.js +270 -0
  48. package/lib/AWS/SageMaker/ClusterSchedulerConfig.js.map +1 -0
  49. package/lib/AWS/SageMaker/ComputeQuota.d.ts +106 -0
  50. package/lib/AWS/SageMaker/ComputeQuota.d.ts.map +1 -0
  51. package/lib/AWS/SageMaker/ComputeQuota.js +268 -0
  52. package/lib/AWS/SageMaker/ComputeQuota.js.map +1 -0
  53. package/lib/AWS/SageMaker/index.d.ts +3 -0
  54. package/lib/AWS/SageMaker/index.d.ts.map +1 -1
  55. package/lib/AWS/SageMaker/index.js +3 -0
  56. package/lib/AWS/SageMaker/index.js.map +1 -1
  57. package/lib/Bundle/PurePlugin.d.ts +29 -7
  58. package/lib/Bundle/PurePlugin.d.ts.map +1 -1
  59. package/lib/Bundle/PurePlugin.js +56 -33
  60. package/lib/Bundle/PurePlugin.js.map +1 -1
  61. package/lib/Cli/commands/nuke.d.ts.map +1 -1
  62. package/lib/Cli/commands/nuke.js +231 -37
  63. package/lib/Cli/commands/nuke.js.map +1 -1
  64. package/lib/Cloudflare/StateStore/Api.d.ts +1 -1
  65. package/lib/Cloudflare/Workers/Assets.d.ts +26 -8
  66. package/lib/Cloudflare/Workers/Assets.d.ts.map +1 -1
  67. package/lib/Cloudflare/Workers/Assets.js +46 -1
  68. package/lib/Cloudflare/Workers/Assets.js.map +1 -1
  69. package/lib/Cloudflare/Workers/LocalWorkerProvider.d.ts +2 -2
  70. package/lib/Cloudflare/Workers/LocalWorkerProvider.d.ts.map +1 -1
  71. package/lib/Cloudflare/Workers/LocalWorkerProvider.js +39 -6
  72. package/lib/Cloudflare/Workers/LocalWorkerProvider.js.map +1 -1
  73. package/lib/Cloudflare/Workers/Worker.d.ts +7 -5
  74. package/lib/Cloudflare/Workers/Worker.d.ts.map +1 -1
  75. package/lib/Cloudflare/Workers/Worker.js +0 -1
  76. package/lib/Cloudflare/Workers/Worker.js.map +1 -1
  77. package/lib/Cloudflare/Workers/WorkerBinding.d.ts +2 -2
  78. package/lib/Cloudflare/Workers/WorkerBundle.d.ts +20 -1
  79. package/lib/Cloudflare/Workers/WorkerBundle.d.ts.map +1 -1
  80. package/lib/Cloudflare/Workers/WorkerBundle.js +11 -0
  81. package/lib/Cloudflare/Workers/WorkerBundle.js.map +1 -1
  82. package/lib/Cloudflare/Workers/WorkerProvider.d.ts.map +1 -1
  83. package/lib/Cloudflare/Workers/WorkerProvider.js +10 -4
  84. package/lib/Cloudflare/Workers/WorkerProvider.js.map +1 -1
  85. package/lib/Cloudflare/Workflows/WorkflowBridge.d.ts +2 -1
  86. package/lib/Cloudflare/Workflows/WorkflowBridge.d.ts.map +1 -1
  87. package/lib/Cloudflare/Workflows/WorkflowBridge.js +5 -5
  88. package/lib/Cloudflare/Workflows/WorkflowBridge.js.map +1 -1
  89. package/lib/Cloudflare/Workflows/index.d.ts +1 -1
  90. package/lib/Cloudflare/Workflows/index.d.ts.map +1 -1
  91. package/lib/Cloudflare/Workflows/index.js +1 -1
  92. package/lib/Cloudflare/Workflows/index.js.map +1 -1
  93. package/lib/Docker/Container.d.ts +26 -0
  94. package/lib/Docker/Container.d.ts.map +1 -1
  95. package/lib/Docker/Container.js +20 -1
  96. package/lib/Docker/Container.js.map +1 -1
  97. package/lib/Docker/Docker.d.ts +2 -0
  98. package/lib/Docker/Docker.d.ts.map +1 -1
  99. package/lib/Docker/Docker.js.map +1 -1
  100. package/lib/Output.d.ts +4 -1
  101. package/lib/Output.d.ts.map +1 -1
  102. package/lib/Output.js.map +1 -1
  103. package/lib/Provider.d.ts +20 -0
  104. package/lib/Provider.d.ts.map +1 -1
  105. package/lib/Provider.js.map +1 -1
  106. package/lib/Random.d.ts +1 -1
  107. package/lib/Random.d.ts.map +1 -1
  108. package/lib/Resource.d.ts +14 -1
  109. package/lib/Resource.d.ts.map +1 -1
  110. package/lib/Resource.js.map +1 -1
  111. package/lib/SQL/D1.d.ts +1 -1
  112. package/lib/SQL/D1.js +1 -1
  113. package/lib/SQL/Postgres.d.ts +1 -1
  114. package/lib/SQL/Postgres.js +1 -1
  115. package/lib/State/HttpStateApi.d.ts +11 -11
  116. package/lib/Test/Alchemy.d.ts +1 -1
  117. package/lib/Test/Alchemy.js +1 -1
  118. package/lib/Test/Vitest.d.ts +58 -0
  119. package/lib/Test/Vitest.d.ts.map +1 -0
  120. package/lib/Test/Vitest.js +114 -0
  121. package/lib/Test/Vitest.js.map +1 -0
  122. package/lib/Util/proxy-chain.d.ts +5 -0
  123. package/lib/Util/proxy-chain.d.ts.map +1 -1
  124. package/lib/Util/proxy-chain.js +52 -4
  125. package/lib/Util/proxy-chain.js.map +1 -1
  126. package/package.json +23 -15
  127. package/src/AWS/EC2/InternetGateway.ts +8 -3
  128. package/src/AWS/EC2/LingeringEnis.ts +169 -0
  129. package/src/AWS/EC2/SecurityGroup.ts +24 -28
  130. package/src/AWS/EC2/Subnet.ts +23 -34
  131. package/src/AWS/EKS/Addon.ts +17 -4
  132. package/src/AWS/EKS/Deployment.ts +26 -5
  133. package/src/AWS/EKS/HyperPod.ts +97 -0
  134. package/src/AWS/EKS/Job.ts +32 -7
  135. package/src/AWS/EKS/internal/client.ts +40 -3
  136. package/src/AWS/Lambda/Function.ts +5 -3
  137. package/src/AWS/Providers.ts +6 -0
  138. package/src/AWS/SageMaker/Cluster.ts +775 -0
  139. package/src/AWS/SageMaker/ClusterSchedulerConfig.ts +430 -0
  140. package/src/AWS/SageMaker/ComputeQuota.ts +426 -0
  141. package/src/AWS/SageMaker/index.ts +3 -0
  142. package/src/Bundle/PurePlugin.ts +92 -39
  143. package/src/Cli/commands/nuke.ts +295 -56
  144. package/src/Cloudflare/Workers/Assets.ts +53 -1
  145. package/src/Cloudflare/Workers/LocalWorkerProvider.ts +46 -7
  146. package/src/Cloudflare/Workers/Worker.ts +7 -5
  147. package/src/Cloudflare/Workers/WorkerBundle.ts +31 -1
  148. package/src/Cloudflare/Workers/WorkerProvider.ts +18 -4
  149. package/src/Cloudflare/Workflows/WorkflowBridge.ts +5 -5
  150. package/src/Cloudflare/Workflows/index.ts +1 -1
  151. package/src/Docker/Container.ts +30 -1
  152. package/src/Docker/Docker.ts +2 -0
  153. package/src/Output.ts +19 -11
  154. package/src/Provider.ts +20 -0
  155. package/src/Resource.ts +18 -1
  156. package/src/SQL/D1.ts +1 -1
  157. package/src/SQL/Postgres.ts +1 -1
  158. package/src/Test/Alchemy.ts +1 -1
  159. package/src/Test/Vitest.ts +244 -0
  160. package/src/Util/proxy-chain.ts +73 -4
  161. package/lib/SQL/index.d.ts +0 -4
  162. package/lib/SQL/index.d.ts.map +0 -1
  163. package/lib/SQL/index.js +0 -4
  164. package/lib/SQL/index.js.map +0 -1
  165. package/src/SQL/index.ts +0 -3
@@ -0,0 +1,169 @@
1
+ import * as ec2 from "@distilled.cloud/aws/ec2";
2
+ import * as Effect from "effect/Effect";
3
+ import * as Result from "effect/Result";
4
+
5
+ /**
6
+ * Shared teardown accelerator for network interfaces that AWS releases
7
+ * *asynchronously* after their owning resource is deleted.
8
+ *
9
+ * The canonical case is a VPC-attached Lambda Function: `DeleteFunction`
10
+ * returns immediately, but the function's Hyperplane ENIs stay `in-use` in
11
+ * their subnets/security groups for several minutes, and Lambda's own reaper
12
+ * may not delete the detached ENI for up to ~20 minutes. Until the ENI is
13
+ * gone, `DeleteSubnet`/`DeleteSecurityGroup` fail with `DependencyViolation`
14
+ * — historically forcing users to run `destroy` twice, minutes apart.
15
+ *
16
+ * `retryWhileLingeringEnis` wraps a subnet/security-group delete call:
17
+ * on every `DependencyViolation` it observes the ENIs still occupying the
18
+ * resource, explicitly deletes any *detached* (status `available`) Lambda
19
+ * ENI instead of waiting for Lambda's reaper, and — while the blockers are
20
+ * exclusively Lambda ENIs that are provably on their way out — keeps waiting
21
+ * on an extended budget that covers AWS's worst-case release window. Any
22
+ * other lingering dependency keeps today's shorter budget.
23
+ */
24
+
25
+ /**
26
+ * ENIs that their owning service tears down asynchronously and that are safe
27
+ * for us to delete once detached. Only Lambda Hyperplane ENIs qualify today:
28
+ * every other managed type (NAT gateways, VPC endpoints, ELB, EFS mount
29
+ * targets, ECS task ENIs) is deleted by the API call that deletes its owner
30
+ * — and their owners are proper alchemy resources ordered ahead of the
31
+ * subnet/SG by the deletion graph.
32
+ *
33
+ * Matching note (observed live): while attached, a Hyperplane ENI reports
34
+ * `InterfaceType: "lambda"` — but the moment Lambda detaches it, the type
35
+ * flips to plain `"interface"` and only the description (`AWS Lambda VPC
36
+ * ENI-{functionName}`) still identifies it. Match both, like Terraform does.
37
+ */
38
+ const isReapableEni = (eni: ec2.NetworkInterface): boolean =>
39
+ eni.InterfaceType === "lambda" ||
40
+ (eni.Description?.startsWith("AWS Lambda VPC ENI") ?? false);
41
+
42
+ export interface LingeringEniScope {
43
+ /** ENI filter naming the resource being deleted. */
44
+ readonly name: "subnet-id" | "group-id";
45
+ readonly value: string;
46
+ }
47
+
48
+ /**
49
+ * Observe ENIs still occupying the resource and delete any detached
50
+ * reapable ones. Best-effort by design: the reaper only accelerates the
51
+ * outer DependencyViolation retry, so an account that cannot describe or
52
+ * delete ENIs degrades to plain waiting instead of failing the destroy.
53
+ */
54
+ const reapLingeringEnis = Effect.fn(function* (
55
+ scope: LingeringEniScope,
56
+ session: { note: (note: string) => Effect.Effect<void> },
57
+ ) {
58
+ const described = yield* ec2
59
+ .describeNetworkInterfaces({
60
+ Filters: [{ Name: scope.name, Values: [scope.value] }],
61
+ })
62
+ .pipe(Effect.catch(() => Effect.succeed({ NetworkInterfaces: [] })));
63
+
64
+ const lingering = (described.NetworkInterfaces ?? []).filter(isReapableEni);
65
+
66
+ let reaped = 0;
67
+ let pending = 0;
68
+ for (const eni of lingering) {
69
+ if (eni.Status !== "available" || eni.NetworkInterfaceId === undefined) {
70
+ pending += 1;
71
+ continue;
72
+ }
73
+ const outcome = yield* ec2
74
+ .deleteNetworkInterface({
75
+ NetworkInterfaceId: eni.NetworkInterfaceId,
76
+ })
77
+ .pipe(
78
+ Effect.as("deleted" as const),
79
+ // Already gone counts as progress; anything else (raced back to
80
+ // in-use, throttle, missing IAM permission) must not fail the
81
+ // destroy — reaping is purely an accelerator, so degrade to the
82
+ // outer retry's plain waiting.
83
+ Effect.catchTag("InvalidNetworkInterfaceID.NotFound", () =>
84
+ Effect.succeed("deleted" as const),
85
+ ),
86
+ Effect.catch(() => Effect.succeed("pending" as const)),
87
+ );
88
+ if (outcome === "deleted") {
89
+ yield* session.note(
90
+ `Deleted detached Lambda ENI ${eni.NetworkInterfaceId}`,
91
+ );
92
+ reaped += 1;
93
+ } else {
94
+ pending += 1;
95
+ }
96
+ }
97
+
98
+ return {
99
+ /** Reapable ENIs still occupying the resource after this pass. */
100
+ pendingReapable: pending,
101
+ /** Detached reapable ENIs actually deleted this pass. */
102
+ reaped,
103
+ };
104
+ });
105
+
106
+ /**
107
+ * Retry `deleteCall` while it fails with a dependency violation, reaping
108
+ * lingering ENIs between attempts.
109
+ *
110
+ * - Base budget (~12 min) matches the historical subnet schedule: fast
111
+ * exponential start, capped at 30-second steps.
112
+ * - While the observed blockers include reapable ENIs (attached or just
113
+ * reaped), the budget extends to ~25 min — Lambda's documented worst-case
114
+ * ENI release window — because those blockers are guaranteed to clear.
115
+ */
116
+ export const retryWhileLingeringEnis = <A, E extends { _tag: string }, R>(
117
+ deleteCall: Effect.Effect<A, E, R>,
118
+ options: {
119
+ scope: LingeringEniScope;
120
+ isDependencyViolation: (error: E) => boolean;
121
+ session: { note: (note: string) => Effect.Effect<void> };
122
+ },
123
+ ) =>
124
+ Effect.gen(function* () {
125
+ const BASE_BUDGET_MILLIS = 12 * 60 * 1000;
126
+ const EXTENDED_BUDGET_MILLIS = 25 * 60 * 1000;
127
+
128
+ let elapsedMillis = 0;
129
+ let sawReapable = false;
130
+
131
+ for (let attempt = 1; ; attempt++) {
132
+ const result = yield* Effect.result(deleteCall);
133
+ if (Result.isSuccess(result)) {
134
+ return result.success;
135
+ }
136
+ const error = result.failure;
137
+ if (!options.isDependencyViolation(error)) {
138
+ return yield* Effect.fail(error);
139
+ }
140
+
141
+ const { pendingReapable, reaped } = yield* reapLingeringEnis(
142
+ options.scope,
143
+ options.session,
144
+ );
145
+ if (reaped > 0) {
146
+ // An ENI just left — the next attempt has a real chance; skip the
147
+ // backoff sleep and try immediately.
148
+ continue;
149
+ }
150
+ sawReapable ||= pendingReapable > 0;
151
+
152
+ const budget = sawReapable ? EXTENDED_BUDGET_MILLIS : BASE_BUDGET_MILLIS;
153
+ if (elapsedMillis >= budget) {
154
+ return yield* Effect.fail(error);
155
+ }
156
+
157
+ yield* options.session.note(
158
+ pendingReapable > 0
159
+ ? `Waiting for AWS to release ${pendingReapable} lingering ENI(s)... (attempt ${attempt})`
160
+ : `Waiting for dependencies to clear... (attempt ${attempt})`,
161
+ );
162
+
163
+ // Fast exponential start capped at 30-second steps (the historical
164
+ // subnet schedule).
165
+ const delayMillis = Math.min(1000 * 1.5 ** (attempt - 1), 30_000);
166
+ yield* Effect.sleep(delayMillis);
167
+ elapsedMillis += delayMillis;
168
+ }
169
+ });
@@ -11,6 +11,7 @@ import type { AccountID } from "../Environment.ts";
11
11
  import { AWSEnvironment } from "../Environment.ts";
12
12
  import type { Providers } from "../Providers.ts";
13
13
  import type { RegionID } from "../Region.ts";
14
+ import { retryWhileLingeringEnis } from "./LingeringEnis.ts";
14
15
  import type { VpcId } from "./Vpc.ts";
15
16
 
16
17
  export type SecurityGroupId<ID extends string = string> = `sg-${ID}`;
@@ -681,34 +682,29 @@ export const SecurityGroupProvider = () =>
681
682
 
682
683
  yield* session.note(`Deleting Security Group: ${groupId}`);
683
684
 
684
- yield* ec2
685
- .deleteSecurityGroup({
686
- GroupId: groupId,
687
- DryRun: false,
688
- })
689
- .pipe(
690
- Effect.catchTag("InvalidGroup.NotFound", () => Effect.void),
691
- // Retry on dependency violations (e.g., ENIs still using the security group)
692
- Effect.retry({
693
- while: (e) => {
694
- return (
695
- e._tag === "DependencyViolation" ||
696
- (e._tag === "ValidationError" &&
697
- e.message?.includes("DependencyViolation"))
698
- );
699
- },
700
- schedule: Schedule.max([
701
- Schedule.fixed(5000),
702
- Schedule.recurs(30),
703
- ]).pipe(
704
- Schedule.tap(({ attempt }) =>
705
- session.note(
706
- `Waiting for dependencies to clear... (attempt ${attempt})`,
707
- ),
708
- ),
709
- ),
710
- }),
711
- );
685
+ // DependencyViolation means ENIs still reference the group — ALB,
686
+ // ECS task, or VPC-attached Lambda ENIs release minutes after the
687
+ // owning resource is deleted. Lambda Hyperplane ENIs are reaped
688
+ // explicitly between attempts (they otherwise linger up to ~20
689
+ // minutes and used to force a second `destroy` run).
690
+ yield* retryWhileLingeringEnis(
691
+ ec2
692
+ .deleteSecurityGroup({
693
+ GroupId: groupId,
694
+ DryRun: false,
695
+ })
696
+ .pipe(
697
+ Effect.catchTag("InvalidGroup.NotFound", () => Effect.void),
698
+ ),
699
+ {
700
+ scope: { name: "group-id", value: groupId },
701
+ isDependencyViolation: (e) =>
702
+ e._tag === "DependencyViolation" ||
703
+ (e._tag === "ValidationError" &&
704
+ (e.message?.includes("DependencyViolation") ?? false)),
705
+ session,
706
+ },
707
+ );
712
708
 
713
709
  yield* session.note(`Security Group ${groupId} deleted`);
714
710
  }),
@@ -7,6 +7,7 @@ import * as Stream from "effect/Stream";
7
7
 
8
8
  import type { ScopedPlanStatusSession } from "../../Cli/Cli.ts";
9
9
  import { isResolved, somePropsAreDifferent } from "../../Diff.ts";
10
+ import { retryWhileLingeringEnis } from "./LingeringEnis.ts";
10
11
  import * as Provider from "../../Provider.ts";
11
12
  import { Resource } from "../../Resource.ts";
12
13
  import type { Providers } from "../Providers.ts";
@@ -597,40 +598,28 @@ export const SubnetProvider = () =>
597
598
 
598
599
  yield* session.note(`Deleting subnet: ${subnetId}`);
599
600
 
600
- // 1. Attempt to delete subnet
601
- yield* ec2
602
- .deleteSubnet({
603
- SubnetId: subnetId,
604
- DryRun: false,
605
- })
606
- .pipe(
607
- Effect.tapError(Effect.logDebug),
608
- Effect.catchTag("InvalidSubnetID.NotFound", () => Effect.void),
609
- // Retry on dependency violations (resources still being deleted).
610
- // ENIs from a just-deleted ALB or CloudFront VPC origin can take
611
- // several minutes to detach after the owning resource is gone, so
612
- // budget ~12 min (fast exponential start, capped at 30s steps).
613
- Effect.retry({
614
- while: (e) => {
615
- // DependencyViolation means there are still dependent resources
616
- // This can happen if ENIs/instances are being deleted concurrently
617
- return e._tag === "DependencyViolation";
618
- },
619
- schedule: Schedule.max([
620
- Schedule.min([
621
- Schedule.exponential(1000, 1.5),
622
- Schedule.spaced("30 seconds"),
623
- ]),
624
- Schedule.recurs(30),
625
- ]).pipe(
626
- Schedule.tap(({ attempt }) =>
627
- session.note(
628
- `Waiting for dependencies to clear... (attempt ${attempt})`,
629
- ),
630
- ),
631
- ),
632
- }),
633
- );
601
+ // 1. Attempt to delete subnet. DependencyViolation means resources
602
+ // still occupy it — ENIs from a just-deleted ALB, CloudFront VPC
603
+ // origin, or a VPC-attached Lambda can take minutes to release
604
+ // after the owning resource is gone. Lambda Hyperplane ENIs are
605
+ // reaped explicitly between attempts (they otherwise linger up to
606
+ // ~20 minutes and used to force a second `destroy` run).
607
+ yield* retryWhileLingeringEnis(
608
+ ec2
609
+ .deleteSubnet({
610
+ SubnetId: subnetId,
611
+ DryRun: false,
612
+ })
613
+ .pipe(
614
+ Effect.tapError(Effect.logDebug),
615
+ Effect.catchTag("InvalidSubnetID.NotFound", () => Effect.void),
616
+ ),
617
+ {
618
+ scope: { name: "subnet-id", value: subnetId },
619
+ isDependencyViolation: (e) => e._tag === "DependencyViolation",
620
+ session,
621
+ },
622
+ );
634
623
 
635
624
  // 2. Wait for subnet to be fully deleted
636
625
  yield* waitForSubnetDeleted(subnetId, session);
@@ -161,7 +161,11 @@ class AddonNotReady extends Data.TaggedError("AddonNotReady")<{
161
161
  readonly clusterName: string;
162
162
  readonly addonName: string;
163
163
  readonly status: string | undefined;
164
- }> {}
164
+ }> {
165
+ override get message(): string {
166
+ return `addon '${this.clusterName}/${this.addonName}' is not ACTIVE (status: ${this.status ?? "absent"})`;
167
+ }
168
+ }
165
169
 
166
170
  class AddonStillExists extends Data.TaggedError("AddonStillExists")<{
167
171
  readonly clusterName: string;
@@ -233,8 +237,14 @@ export const AddonProvider = () =>
233
237
  }
234
238
  }),
235
239
  read: Effect.fn(function* ({ id, olds }) {
240
+ const clusterName = olds.clusterName as string | undefined;
241
+ // A crashed prior run can persist a row before its unresolved
242
+ // inputs were stripped — nothing observable yet.
243
+ if (clusterName === undefined || olds.addonName === undefined) {
244
+ return undefined;
245
+ }
236
246
  const state = yield* readAddon({
237
- clusterName: olds.clusterName as string,
247
+ clusterName,
238
248
  addonName: olds.addonName,
239
249
  });
240
250
  if (!state) return undefined;
@@ -404,11 +414,14 @@ const waitForAddonActive = Effect.fn(function* ({
404
414
  }),
405
415
  Effect.retry({
406
416
  while: (error) => error instanceof AddonNotReady,
407
- // Flat 5s polls, ~10 min budget (uncapped exponential sleeps for
417
+ // Flat 5s polls, ~20 min budget (uncapped exponential sleeps for
408
418
  // multi-minute stretches late in the wait — looks like a deadlock).
419
+ // Node-bound addons (e.g. HyperPod task governance) stay DEGRADED
420
+ // until nodes join and pull images, which can take most of the
421
+ // budget when the addon installs alongside its node group.
409
422
  schedule: Schedule.max([
410
423
  Schedule.spaced("5 seconds"),
411
- Schedule.recurs(120),
424
+ Schedule.recurs(240),
412
425
  ]),
413
426
  }),
414
427
  );
@@ -35,6 +35,12 @@ import {
35
35
  deleteObjects,
36
36
  type KubernetesClusterConnection,
37
37
  } from "./internal/client.ts";
38
+ import {
39
+ hyperpodNamespace,
40
+ hyperpodNodeSelector,
41
+ hyperpodWorkloadLabels,
42
+ type HyperPodWorkloadProps,
43
+ } from "./HyperPod.ts";
38
44
  import {
39
45
  toKubernetesObjectRef,
40
46
  type KubernetesObjectDefinition,
@@ -76,9 +82,17 @@ export interface DeploymentPropsBase extends PlatformProps {
76
82
  /**
77
83
  * Kubernetes namespace to deploy into. The namespace must already exist
78
84
  * (Auto Mode clusters ship a `default` namespace).
79
- * @default "default"
85
+ * @default "default" (or `hyperpod-ns-<team>` when `hyperpod.quota` is set)
80
86
  */
81
87
  namespace?: string;
88
+ /**
89
+ * Run on SageMaker HyperPod nodes attached to this EKS cluster: pin to an
90
+ * instance group, keep off unhealthy nodes, and optionally submit through
91
+ * HyperPod task governance by passing the team's
92
+ * `AWS.SageMaker.ComputeQuota` (which derives the namespace and Kueue
93
+ * labels).
94
+ */
95
+ hyperpod?: HyperPodWorkloadProps;
82
96
  /**
83
97
  * HTTP port exposed by the container and the Service.
84
98
  * @default 3000
@@ -539,9 +553,11 @@ export const DeploymentProvider = () =>
539
553
  ) {
540
554
  return { action: "replace" } as const;
541
555
  }
556
+ const effectiveNamespace = (props: DeploymentProps) =>
557
+ props.namespace ?? hyperpodNamespace(props.hyperpod) ?? "default";
542
558
  if (
543
- olds.namespace !== undefined &&
544
- (olds.namespace ?? "default") !== (news.namespace ?? "default")
559
+ olds.cluster?.clusterName !== undefined &&
560
+ effectiveNamespace(olds) !== effectiveNamespace(news)
545
561
  ) {
546
562
  return { action: "replace" } as const;
547
563
  }
@@ -585,7 +601,8 @@ export const DeploymentProvider = () =>
585
601
  }) {
586
602
  const connection = toConnection(news.cluster);
587
603
  const clusterName = news.cluster.clusterName;
588
- const namespace = news.namespace ?? "default";
604
+ const namespace =
605
+ news.namespace ?? hyperpodNamespace(news.hyperpod) ?? "default";
589
606
  const port = news.port ?? 3000;
590
607
  const serviceType = news.serviceType ?? "LoadBalancer";
591
608
 
@@ -651,7 +668,10 @@ export const DeploymentProvider = () =>
651
668
  // get no region env var — inject it so the bootstrap's
652
669
  // `Region.fromEnv()` resolves inside the pod.
653
670
  const { region } = yield* AWSEnvironment.current;
654
- const labels = news.labels ?? { "app.kubernetes.io/name": baseName };
671
+ const labels = {
672
+ ...(news.labels ?? { "app.kubernetes.io/name": baseName }),
673
+ ...hyperpodWorkloadLabels(news.hyperpod),
674
+ };
655
675
  const containerEnv = {
656
676
  ...bindingEnv,
657
677
  ...alchemyEnv,
@@ -670,6 +690,7 @@ export const DeploymentProvider = () =>
670
690
  metadata: { labels },
671
691
  spec: {
672
692
  serviceAccountName,
693
+ nodeSelector: hyperpodNodeSelector(news.hyperpod),
673
694
  containers: [
674
695
  {
675
696
  name: baseName,
@@ -0,0 +1,97 @@
1
+ /**
2
+ * First-class SageMaker HyperPod scheduling for EKS workloads.
3
+ *
4
+ * HyperPod nodes are ordinary EKS nodes carrying well-known labels, and
5
+ * HyperPod task governance rides on Kueue conventions. `AWS.EKS.Job` and
6
+ * `AWS.EKS.Deployment` accept these props under `hyperpod:` and derive the
7
+ * node selector, namespace, and Kueue labels — no manual label wiring.
8
+ */
9
+
10
+ /** The well-known node label carrying the HyperPod instance-group name. */
11
+ export const HYPERPOD_INSTANCE_GROUP_LABEL =
12
+ "sagemaker.amazonaws.com/instance-group-name";
13
+
14
+ /** The well-known node label carrying HyperPod's node health verdict. */
15
+ export const HYPERPOD_NODE_HEALTH_LABEL =
16
+ "sagemaker.amazonaws.com/node-health-status";
17
+
18
+ /** Kueue label selecting the task-governance queue. */
19
+ export const KUEUE_QUEUE_NAME_LABEL = "kueue.x-k8s.io/queue-name";
20
+
21
+ /** Kueue label selecting the task-governance priority class. */
22
+ export const KUEUE_PRIORITY_CLASS_LABEL = "kueue.x-k8s.io/priority-class";
23
+
24
+ export interface HyperPodWorkloadProps {
25
+ /**
26
+ * Pin the workload to a HyperPod instance group (matches the
27
+ * `sagemaker.amazonaws.com/instance-group-name` node label). Reference
28
+ * the group through the cluster's attributes —
29
+ * `hyperpod.instanceGroups.workers` — so the workload is connected to
30
+ * the cluster through the resource graph. A plain name string also
31
+ * works.
32
+ */
33
+ instanceGroup?: string | { InstanceGroupName?: string };
34
+ /**
35
+ * Only schedule onto nodes that passed HyperPod health checks
36
+ * (`sagemaker.amazonaws.com/node-health-status: Schedulable`).
37
+ * @default true
38
+ */
39
+ healthyNodesOnly?: boolean;
40
+ /**
41
+ * Submit through HyperPod task governance under this team's quota — pass
42
+ * the `AWS.SageMaker.ComputeQuota` resource. Derives the
43
+ * `hyperpod-ns-<team>` namespace and the Kueue queue label (both
44
+ * materialized by the quota), and orders the workload after it.
45
+ */
46
+ quota?: {
47
+ /** The quota's team name (`ComputeQuota.teamName`). */
48
+ teamName: string;
49
+ };
50
+ /**
51
+ * The task-governance priority class (a `PriorityClass` name from the
52
+ * cluster's `AWS.SageMaker.ClusterSchedulerConfig`).
53
+ */
54
+ priorityClass?: string;
55
+ }
56
+
57
+ /** @internal The `hyperpod-ns-<team>` namespace, when governed. */
58
+ export const hyperpodNamespace = (
59
+ hyperpod: HyperPodWorkloadProps | undefined,
60
+ ): string | undefined =>
61
+ hyperpod?.quota !== undefined
62
+ ? `hyperpod-ns-${hyperpod.quota.teamName}`
63
+ : undefined;
64
+
65
+ /** @internal Kueue labels for the workload object. */
66
+ export const hyperpodWorkloadLabels = (
67
+ hyperpod: HyperPodWorkloadProps | undefined,
68
+ ): Record<string, string> => ({
69
+ ...(hyperpod?.quota !== undefined
70
+ ? {
71
+ [KUEUE_QUEUE_NAME_LABEL]: `hyperpod-ns-${hyperpod.quota.teamName}-localqueue`,
72
+ }
73
+ : {}),
74
+ ...(hyperpod?.priorityClass !== undefined
75
+ ? { [KUEUE_PRIORITY_CLASS_LABEL]: `${hyperpod.priorityClass}-priority` }
76
+ : {}),
77
+ });
78
+
79
+ /** @internal The instance-group name from either reference form. */
80
+ const instanceGroupName = (
81
+ group: string | { InstanceGroupName?: string } | undefined,
82
+ ): string | undefined =>
83
+ typeof group === "string" ? group : group?.InstanceGroupName;
84
+
85
+ /** @internal Node selector pinning pods onto HyperPod nodes. */
86
+ export const hyperpodNodeSelector = (
87
+ hyperpod: HyperPodWorkloadProps | undefined,
88
+ ): Record<string, string> | undefined => {
89
+ if (hyperpod === undefined) return undefined;
90
+ const group = instanceGroupName(hyperpod.instanceGroup);
91
+ return {
92
+ ...(hyperpod.healthyNodesOnly !== false
93
+ ? { [HYPERPOD_NODE_HEALTH_LABEL]: "Schedulable" }
94
+ : {}),
95
+ ...(group !== undefined ? { [HYPERPOD_INSTANCE_GROUP_LABEL]: group } : {}),
96
+ };
97
+ };
@@ -35,6 +35,12 @@ import {
35
35
  import { AWSEnvironment } from "../Environment.ts";
36
36
  import type { PolicyStatement } from "../IAM/Policy.ts";
37
37
  import type { Providers } from "../Providers.ts";
38
+ import {
39
+ hyperpodNamespace,
40
+ hyperpodNodeSelector,
41
+ hyperpodWorkloadLabels,
42
+ type HyperPodWorkloadProps,
43
+ } from "./HyperPod.ts";
38
44
  import { reconcileObjects, deleteObjects } from "./internal/client.ts";
39
45
  import type {
40
46
  KubernetesObjectDefinition,
@@ -75,9 +81,17 @@ export interface JobPropsBase extends PlatformProps {
75
81
  /**
76
82
  * Kubernetes namespace to run in. The namespace must already exist (Auto
77
83
  * Mode clusters ship a `default` namespace).
78
- * @default "default"
84
+ * @default "default" (or `hyperpod-ns-<team>` when `hyperpod.quota` is set)
79
85
  */
80
86
  namespace?: string;
87
+ /**
88
+ * Run on SageMaker HyperPod nodes attached to this EKS cluster: pin to an
89
+ * instance group, keep off unhealthy nodes, and optionally submit through
90
+ * HyperPod task governance by passing the team's
91
+ * `AWS.SageMaker.ComputeQuota` (which derives the namespace and Kueue
92
+ * labels).
93
+ */
94
+ hyperpod?: HyperPodWorkloadProps;
81
95
  /**
82
96
  * Number of retries before the Job is marked failed (Kubernetes
83
97
  * `backoffLimit`).
@@ -472,9 +486,11 @@ export const JobProvider = () =>
472
486
  ) {
473
487
  return { action: "replace" } as const;
474
488
  }
489
+ const effectiveNamespace = (props: JobProps) =>
490
+ props.namespace ?? hyperpodNamespace(props.hyperpod) ?? "default";
475
491
  if (
476
- olds.namespace !== undefined &&
477
- (olds.namespace ?? "default") !== (news.namespace ?? "default")
492
+ olds.cluster?.clusterName !== undefined &&
493
+ effectiveNamespace(olds) !== effectiveNamespace(news)
478
494
  ) {
479
495
  return { action: "replace" } as const;
480
496
  }
@@ -515,7 +531,8 @@ export const JobProvider = () =>
515
531
  }) {
516
532
  const connection = toConnection(news.cluster);
517
533
  const clusterName = news.cluster.clusterName;
518
- const namespace = news.namespace ?? "default";
534
+ const namespace =
535
+ news.namespace ?? hyperpodNamespace(news.hyperpod) ?? "default";
519
536
 
520
537
  const baseName = yield* toBaseName(id, news);
521
538
  const serviceAccountName = output?.serviceAccountName ?? baseName;
@@ -569,7 +586,10 @@ export const JobProvider = () =>
569
586
  });
570
587
 
571
588
  // Synthesize the Kubernetes objects.
572
- const labels = news.labels ?? { "app.kubernetes.io/name": baseName };
589
+ const labels = {
590
+ ...(news.labels ?? { "app.kubernetes.io/name": baseName }),
591
+ ...hyperpodWorkloadLabels(news.hyperpod),
592
+ };
573
593
  const containerEnv = {
574
594
  ...bindingEnv,
575
595
  ...alchemyEnv,
@@ -586,6 +606,7 @@ export const JobProvider = () =>
586
606
  metadata: { labels },
587
607
  spec: {
588
608
  serviceAccountName,
609
+ nodeSelector: hyperpodNodeSelector(news.hyperpod),
589
610
  restartPolicy: news.restartPolicy ?? "Never",
590
611
  containers: [
591
612
  {
@@ -620,9 +641,13 @@ export const JobProvider = () =>
620
641
  // hash of the spec, and a spec change applies a NEW Job (which
621
642
  // runs) while `reconcileObjects` deletes the previous one.
622
643
  // CronJobs are mutable and keep the stable base name.
644
+ // Kubernetes rejects Job names over 63 characters (the API
645
+ // stamps the name into the batch.kubernetes.io/job-name pod
646
+ // label, and label values cap at 63) — truncate the base so the
647
+ // content-address suffix always fits.
623
648
  const jobName = news.schedule
624
- ? baseName
625
- : `${baseName}-${(yield* sha256Object(jobSpec)).slice(0, 8)}`;
649
+ ? baseName.slice(0, 52).replace(/-+$/, "")
650
+ : `${baseName.slice(0, 54).replace(/-+$/, "")}-${(yield* sha256Object(jobSpec)).slice(0, 8)}`;
626
651
 
627
652
  const workloadObject: KubernetesObjectDefinition = news.schedule
628
653
  ? {