@tickernelz/paperclip-pro-plugin-kubernetes 2026.925.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (100) hide show
  1. package/LICENSE +22 -0
  2. package/README.md +180 -0
  3. package/dist/adapter-defaults.d.ts +29 -0
  4. package/dist/adapter-defaults.d.ts.map +1 -0
  5. package/dist/adapter-defaults.js +98 -0
  6. package/dist/adapter-defaults.js.map +1 -0
  7. package/dist/adapter-registry.d.ts +64 -0
  8. package/dist/adapter-registry.d.ts.map +1 -0
  9. package/dist/adapter-registry.js +25 -0
  10. package/dist/adapter-registry.js.map +1 -0
  11. package/dist/cilium-network-policy.d.ts +12 -0
  12. package/dist/cilium-network-policy.d.ts.map +1 -0
  13. package/dist/cilium-network-policy.js +60 -0
  14. package/dist/cilium-network-policy.js.map +1 -0
  15. package/dist/file-sync.d.ts +73 -0
  16. package/dist/file-sync.d.ts.map +1 -0
  17. package/dist/file-sync.js +752 -0
  18. package/dist/file-sync.js.map +1 -0
  19. package/dist/image-allowlist.d.ts +18 -0
  20. package/dist/image-allowlist.d.ts.map +1 -0
  21. package/dist/image-allowlist.js +39 -0
  22. package/dist/image-allowlist.js.map +1 -0
  23. package/dist/index.d.ts +3 -0
  24. package/dist/index.d.ts.map +1 -0
  25. package/dist/index.js +3 -0
  26. package/dist/index.js.map +1 -0
  27. package/dist/job-orchestrator.d.ts +24 -0
  28. package/dist/job-orchestrator.d.ts.map +1 -0
  29. package/dist/job-orchestrator.js +90 -0
  30. package/dist/job-orchestrator.js.map +1 -0
  31. package/dist/kube-client.d.ts +15 -0
  32. package/dist/kube-client.d.ts.map +1 -0
  33. package/dist/kube-client.js +23 -0
  34. package/dist/kube-client.js.map +1 -0
  35. package/dist/lease-lifecycle.d.ts +62 -0
  36. package/dist/lease-lifecycle.d.ts.map +1 -0
  37. package/dist/lease-lifecycle.js +155 -0
  38. package/dist/lease-lifecycle.js.map +1 -0
  39. package/dist/manifest.d.ts +4 -0
  40. package/dist/manifest.d.ts.map +1 -0
  41. package/dist/manifest.js +108 -0
  42. package/dist/manifest.js.map +1 -0
  43. package/dist/network-policy.d.ts +22 -0
  44. package/dist/network-policy.d.ts.map +1 -0
  45. package/dist/network-policy.js +116 -0
  46. package/dist/network-policy.js.map +1 -0
  47. package/dist/plugin.d.ts +3 -0
  48. package/dist/plugin.d.ts.map +1 -0
  49. package/dist/plugin.js +820 -0
  50. package/dist/plugin.js.map +1 -0
  51. package/dist/pod-exec.d.ts +67 -0
  52. package/dist/pod-exec.d.ts.map +1 -0
  53. package/dist/pod-exec.js +401 -0
  54. package/dist/pod-exec.js.map +1 -0
  55. package/dist/pod-spec-builder.d.ts +25 -0
  56. package/dist/pod-spec-builder.d.ts.map +1 -0
  57. package/dist/pod-spec-builder.js +82 -0
  58. package/dist/pod-spec-builder.js.map +1 -0
  59. package/dist/sandbox-cr-builder.d.ts +40 -0
  60. package/dist/sandbox-cr-builder.d.ts.map +1 -0
  61. package/dist/sandbox-cr-builder.js +123 -0
  62. package/dist/sandbox-cr-builder.js.map +1 -0
  63. package/dist/sandbox-cr-orchestrator.d.ts +62 -0
  64. package/dist/sandbox-cr-orchestrator.d.ts.map +1 -0
  65. package/dist/sandbox-cr-orchestrator.js +248 -0
  66. package/dist/sandbox-cr-orchestrator.js.map +1 -0
  67. package/dist/sandbox-orchestrator.d.ts +39 -0
  68. package/dist/sandbox-orchestrator.d.ts.map +1 -0
  69. package/dist/sandbox-orchestrator.js +2 -0
  70. package/dist/sandbox-orchestrator.js.map +1 -0
  71. package/dist/scoped-network-egress.d.ts +19 -0
  72. package/dist/scoped-network-egress.d.ts.map +1 -0
  73. package/dist/scoped-network-egress.js +86 -0
  74. package/dist/scoped-network-egress.js.map +1 -0
  75. package/dist/secret-manager.d.ts +14 -0
  76. package/dist/secret-manager.d.ts.map +1 -0
  77. package/dist/secret-manager.js +39 -0
  78. package/dist/secret-manager.js.map +1 -0
  79. package/dist/tenant-orchestrator.d.ts +35 -0
  80. package/dist/tenant-orchestrator.d.ts.map +1 -0
  81. package/dist/tenant-orchestrator.js +263 -0
  82. package/dist/tenant-orchestrator.js.map +1 -0
  83. package/dist/types.d.ts +279 -0
  84. package/dist/types.d.ts.map +1 -0
  85. package/dist/types.js +65 -0
  86. package/dist/types.js.map +1 -0
  87. package/dist/upload-interceptor.d.ts +44 -0
  88. package/dist/upload-interceptor.d.ts.map +1 -0
  89. package/dist/upload-interceptor.js +114 -0
  90. package/dist/upload-interceptor.js.map +1 -0
  91. package/dist/utils.d.ts +11 -0
  92. package/dist/utils.d.ts.map +1 -0
  93. package/dist/utils.js +51 -0
  94. package/dist/utils.js.map +1 -0
  95. package/dist/worker.d.ts +3 -0
  96. package/dist/worker.d.ts.map +1 -0
  97. package/dist/worker.js +5 -0
  98. package/dist/worker.js.map +1 -0
  99. package/manifests/operator-prerequisites.yaml +22 -0
  100. package/package.json +47 -0
package/dist/plugin.js ADDED
@@ -0,0 +1,820 @@
1
+ import { randomBytes } from "node:crypto";
2
+ import { definePlugin } from "@tickernelz/paperclip-pro-plugin-sdk";
3
+ import { kubernetesProviderConfigSchema, } from "./types.js";
4
+ import { createKubeConfig, makeKubeClients } from "./kube-client.js";
5
+ import { getAdapterDefaults, buildAdapterEnv, resolveRunAdapterType } from "./adapter-defaults.js";
6
+ import { resolveImage } from "./image-allowlist.js";
7
+ import { buildJobManifest } from "./pod-spec-builder.js";
8
+ import { buildSandboxCrManifest } from "./sandbox-cr-builder.js";
9
+ import { ensureTenant } from "./tenant-orchestrator.js";
10
+ import { createPerRunSecret } from "./secret-manager.js";
11
+ import { FastUploadInterceptor } from "./upload-interceptor.js";
12
+ import { jobOrchestrator, JobTimeoutError } from "./job-orchestrator.js";
13
+ import { sandboxCrOrchestrator, SandboxCrTimeoutError, } from "./sandbox-cr-orchestrator.js";
14
+ import { execInPod, execInPodStreaming, wrapCommandWithEnv } from "./pod-exec.js";
15
+ import { performSyncIn, performSyncOut } from "./file-sync.js";
16
+ import { checkLeaseResumable, destroyLeaseResources } from "./lease-lifecycle.js";
17
+ import { appendNetworkEgressDenyHint, createScopedNetworkEgressPolicyOrReleaseWorkload, NETWORK_EGRESS_GRANT_PATH, parseScopedNetworkEgressGrant, } from "./scoped-network-egress.js";
18
+ import { deriveCompanySlug, deriveNamespaceName, newRunUlidDns, paperclipLabels, } from "./utils.js";
19
+ // The namespace paperclip-server itself runs in. Used when building
20
+ // NetworkPolicy manifests so the tenant namespace allows inbound traffic
21
+ // from the server pod.
22
+ const PAPERCLIP_SERVER_NAMESPACE = "paperclip";
23
+ // Name of the ServiceAccount created inside each tenant namespace by ensureTenant.
24
+ const TENANT_SERVICE_ACCOUNT = "paperclip-tenant-sa";
25
+ // Resource quota defaults applied to every tenant namespace (tunable via
26
+ // config in a future iteration).
27
+ const DEFAULT_RESOURCE_QUOTA = {
28
+ pods: "20",
29
+ requestsCpu: "10",
30
+ requestsMemory: "20Gi",
31
+ limitsCpu: "20",
32
+ limitsMemory: "40Gi",
33
+ };
34
+ function deriveTenantNamespace(config, companyId) {
35
+ // TODO: future versions could thread companyName through AcquireLeaseParams
36
+ // to get a friendlier slug (e.g. "acme-corp") instead of the UUID-derived one.
37
+ const slug = config.companySlug ?? deriveCompanySlug(companyId);
38
+ return deriveNamespaceName(config.namespacePrefix, slug);
39
+ }
40
+ function generateBootstrapToken() {
41
+ // TODO: tighten once the agent runtime shim (companion images PR) lands its
42
+ // callback auth scheme; paperclip-server's callback auth is out of scope for
43
+ // this plugin. For now this per-run random token is stored in the per-run
44
+ // Secret and read by the runtime image entrypoint for initial registration.
45
+ return randomBytes(32).toString("hex");
46
+ }
47
+ // One FastUploadInterceptor instance per active lease. Scoping per lease
48
+ // prevents `releaseLease` from wiping in-flight upload buffers belonging to
49
+ // other concurrent leases — a single shared singleton would do exactly that
50
+ // on `reset()`. The Map is keyed by `providerLeaseId`; entries are lazily
51
+ // created in `onEnvironmentExecute` and removed in `onEnvironmentReleaseLease`.
52
+ const uploadInterceptorsByLease = new Map();
53
+ function getOrCreateUploadInterceptor(leaseId) {
54
+ let interceptor = uploadInterceptorsByLease.get(leaseId);
55
+ if (!interceptor) {
56
+ interceptor = new FastUploadInterceptor();
57
+ uploadInterceptorsByLease.set(leaseId, interceptor);
58
+ }
59
+ return interceptor;
60
+ }
61
+ // In-memory cache of sandbox CR names we've already observed reaching the
62
+ // Ready condition during the current plugin-worker lifetime. The k8s
63
+ // sandbox-cr lifecycle means once a Sandbox pod is Running, subsequent
64
+ // execs into it don't need another readiness poll — saves one
65
+ // `getNamespacedCustomObject` round-trip per exec, which adds up across
66
+ // dozens of sequential exec calls in a typical adapter workflow.
67
+ // On worker restart this resets, which is fine: the first exec on each
68
+ // lease then re-confirms readiness from scratch.
69
+ const readySandboxesByLease = new Set();
70
+ // How long onEnvironmentResumeLease waits for an existing Sandbox pod to
71
+ // report Ready before declaring the lease non-resumable. Deliberately short:
72
+ // this is a liveness check on an already-provisioned pod, not a fresh
73
+ // provision — if the pod isn't (almost) up, falling back to acquireLease is
74
+ // faster and more reliable than waiting.
75
+ const RESUME_READY_TIMEOUT_MS = 30_000;
76
+ const RESUME_READY_POLL_MS = 1_000;
77
+ // The workspace remote dir is the confinement root for native file sync. It is
78
+ // recorded on the lease metadata at realizeWorkspace time (`remoteCwd`); require
79
+ // it so a sync can never run without a concrete root to confine every sandbox
80
+ // path against.
81
+ function resolveSyncRemoteDir(lease) {
82
+ const remoteCwd = lease.metadata?.remoteCwd;
83
+ if (typeof remoteCwd === "string" && remoteCwd.trim().length > 0) {
84
+ return remoteCwd.trim();
85
+ }
86
+ throw new Error("Kubernetes file sync requires a workspace remote dir on the lease metadata.");
87
+ }
88
+ /**
89
+ * Resolve the running Sandbox-CR pod for a native file-sync operation and return
90
+ * a `PodStreamExec` bound to it, exactly like `onEnvironmentExecute` resolves its exec
91
+ * target: parse config, derive the namespace, wait for the Sandbox pod to reach
92
+ * Ready (cached per lease), and find the pod name. The `job` backend carries no
93
+ * file path and is out of scope — file sync is only supported on `sandbox-cr`.
94
+ */
95
+ async function resolveSyncPodExec(params) {
96
+ const { lease } = params;
97
+ if (!lease.providerLeaseId) {
98
+ throw new Error("Kubernetes file sync requires a provider lease ID.");
99
+ }
100
+ const config = kubernetesProviderConfigSchema.parse(params.config);
101
+ const namespace = typeof lease.metadata?.namespace === "string"
102
+ ? lease.metadata.namespace
103
+ : deriveTenantNamespace(config, params.companyId);
104
+ const leaseBackend = typeof lease.metadata?.backend === "string"
105
+ ? lease.metadata.backend
106
+ : config.backend;
107
+ if (leaseBackend !== "sandbox-cr") {
108
+ throw new Error(`Kubernetes file sync is only supported on the sandbox-cr backend (lease backend: ${leaseBackend}).`);
109
+ }
110
+ const kc = createKubeConfig({
111
+ inCluster: config.inCluster,
112
+ kubeconfig: config.kubeconfig,
113
+ });
114
+ const clients = makeKubeClients(kc);
115
+ const timeoutMs = config.podActivityDeadlineSec * 1000;
116
+ // Ensure the Sandbox pod is Ready (wait only the first time for this lease),
117
+ // then resolve the pod name — mirrors the onEnvironmentExecute resolution.
118
+ if (!readySandboxesByLease.has(lease.providerLeaseId)) {
119
+ await sandboxCrOrchestrator.waitForCompletion(clients, namespace, lease.providerLeaseId, {
120
+ timeoutMs,
121
+ pollMs: 2000,
122
+ });
123
+ readySandboxesByLease.add(lease.providerLeaseId);
124
+ }
125
+ const podName = typeof lease.metadata?.podName === "string" && lease.metadata.podName
126
+ ? lease.metadata.podName
127
+ : await sandboxCrOrchestrator.findPod(clients, namespace, lease.providerLeaseId);
128
+ if (!podName) {
129
+ throw new Error("Kubernetes file sync could not resolve the Sandbox pod name.");
130
+ }
131
+ // Bind the streaming exec: raw tar bytes move over stdin/stdout straight to and
132
+ // from a host file, so neither side buffers the whole payload. The file-sync
133
+ // module bounds the untrusted pod's stdout with its own streamed-bytes disk
134
+ // guard and passes the stderr cap through `io`.
135
+ const exec = (command, io) => execInPodStreaming(kc, namespace, podName, "agent", command, {
136
+ ...io,
137
+ timeoutMs: io.timeoutMs ?? timeoutMs,
138
+ });
139
+ return { exec, timeoutMs };
140
+ }
141
+ const plugin = definePlugin({
142
+ async setup(ctx) {
143
+ ctx.logger.info("Kubernetes sandbox provider plugin ready");
144
+ },
145
+ async onHealth() {
146
+ return { status: "ok", message: "Kubernetes sandbox provider plugin healthy" };
147
+ },
148
+ async onEnvironmentValidateConfig(params) {
149
+ const parsed = kubernetesProviderConfigSchema.safeParse(params.config);
150
+ if (!parsed.success) {
151
+ return {
152
+ ok: false,
153
+ errors: parsed.error.issues.map((i) => i.message),
154
+ };
155
+ }
156
+ const warnings = [];
157
+ const cfg = parsed.data;
158
+ const adapterDefaults = getAdapterDefaults(cfg.adapterType, cfg.adapters);
159
+ const totalFqdns = [...adapterDefaults.allowFqdns, ...cfg.egressAllowFqdns];
160
+ if (cfg.egressMode === "standard" && totalFqdns.length > 0) {
161
+ if (cfg.egressAllowCidrs.length === 0) {
162
+ warnings.push(`egressMode=standard cannot enforce FQDN-based egress rules (Kubernetes NetworkPolicy is CIDR-only). To keep the configured FQDNs reachable (${totalFqdns.join(", ")}) without operator intervention, the plugin will allow public IPv4 egress on TCP 80/443 with private/link-local/loopback/multicast ranges excluded. This is broader than exact FQDN allow-listing — switch egressMode to "cilium" (requires Cilium CNI) for precise enforcement, or set egressAllowCidrs explicitly to override the fallback.`);
163
+ }
164
+ else {
165
+ warnings.push(`egressMode=standard cannot enforce FQDN-based egress rules. The following FQDNs are reachable only via the operator-supplied egressAllowCidrs: ${totalFqdns.join(", ")}. Switch egressMode to "cilium" (requires Cilium CNI) for exact FQDN allow-listing.`);
166
+ }
167
+ }
168
+ return { ok: true, normalizedConfig: cfg, warnings: warnings.length > 0 ? warnings : undefined };
169
+ },
170
+ async onEnvironmentProbe(params) {
171
+ const parsed = kubernetesProviderConfigSchema.safeParse(params.config);
172
+ if (!parsed.success) {
173
+ return {
174
+ ok: false,
175
+ summary: "Invalid Kubernetes provider configuration.",
176
+ metadata: {
177
+ errors: parsed.error.issues.map((i) => i.message),
178
+ },
179
+ };
180
+ }
181
+ const config = parsed.data;
182
+ const namespace = deriveTenantNamespace(config, params.companyId);
183
+ try {
184
+ const kc = createKubeConfig({
185
+ inCluster: config.inCluster,
186
+ kubeconfig: config.kubeconfig,
187
+ });
188
+ const clients = makeKubeClients(kc);
189
+ // Reachability check: list pods in the tenant namespace. If the namespace
190
+ // doesn't exist yet this will throw a 404 which we treat as "reachable
191
+ // but namespace not provisioned" — still a successful probe.
192
+ try {
193
+ await clients.core.listNamespacedPod({ namespace });
194
+ }
195
+ catch (err) {
196
+ const code = err.code
197
+ ?? err.statusCode;
198
+ if (code !== 404)
199
+ throw err;
200
+ // 404 means namespace doesn't exist yet — cluster is reachable.
201
+ }
202
+ return {
203
+ ok: true,
204
+ summary: `Kubernetes cluster reachable. Tenant namespace: ${namespace}.`,
205
+ metadata: { namespace, provider: "kubernetes" },
206
+ };
207
+ }
208
+ catch (err) {
209
+ return {
210
+ ok: false,
211
+ summary: "Kubernetes cluster probe failed.",
212
+ metadata: {
213
+ namespace,
214
+ provider: "kubernetes",
215
+ error: err instanceof Error ? err.message : String(err),
216
+ },
217
+ };
218
+ }
219
+ },
220
+ async onEnvironmentAcquireLease(
221
+ // `adapterType` is an optional per-run hint the server may pass once the
222
+ // SDK lease params grow that field (companion server-integration PR). The
223
+ // plugin works without it: absent means "use the environment's configured
224
+ // default adapter", so it stays compatible with the current SDK.
225
+ params) {
226
+ const config = kubernetesProviderConfigSchema.parse(params.config);
227
+ const namespace = deriveTenantNamespace(config, params.companyId);
228
+ // The adapter for THIS run is the agent's adapter (params.adapterType) when
229
+ // supplied, so one environment can serve mixed harnesses; otherwise fall back
230
+ // to the environment's configured default adapter. getAdapterDefaults validates
231
+ // it is a registered adapter (throws otherwise), so a curated-out adapter fails
232
+ // the lease as before.
233
+ const effectiveAdapterType = resolveRunAdapterType(params.adapterType, config.adapterType);
234
+ // Emit a runtime warning if FQDNs are configured but egressMode=standard
235
+ // cannot enforce them. Mirrors the validateConfig warning so operators see
236
+ // it in paperclip-server logs even if they missed the validation step.
237
+ const adapterDefaultsForWarn = getAdapterDefaults(effectiveAdapterType, config.adapters);
238
+ const totalFqdnsForWarn = [...adapterDefaultsForWarn.allowFqdns, ...config.egressAllowFqdns];
239
+ if (config.egressMode === "standard" && totalFqdnsForWarn.length > 0) {
240
+ if (config.egressAllowCidrs.length === 0) {
241
+ console.warn(`[plugin-kubernetes] egressMode=standard cannot enforce FQDN-based egress rules; falling back to public-IPv4 (TCP 80/443) with private/link-local ranges excluded so the configured FQDNs (${totalFqdnsForWarn.join(", ")}) remain reachable. Switch egressMode to "cilium" for exact FQDN allow-listing.`);
242
+ }
243
+ else {
244
+ console.warn(`[plugin-kubernetes] egressMode=standard cannot enforce FQDN-based egress rules. The following FQDNs are reachable only via operator-supplied egressAllowCidrs: ${totalFqdnsForWarn.join(", ")}. Switch egressMode to "cilium" for exact FQDN allow-listing.`);
245
+ }
246
+ }
247
+ const kc = createKubeConfig({
248
+ inCluster: config.inCluster,
249
+ kubeconfig: config.kubeconfig,
250
+ });
251
+ const clients = makeKubeClients(kc);
252
+ // Ensure the tenant namespace and all its RBAC / network policy resources
253
+ // exist before we try to create the Job.
254
+ const adapterDefaults = getAdapterDefaults(effectiveAdapterType, config.adapters);
255
+ await ensureTenant(clients, {
256
+ namespace,
257
+ companyId: params.companyId,
258
+ paperclipServerNamespace: PAPERCLIP_SERVER_NAMESPACE,
259
+ serviceAccountAnnotations: config.serviceAccountAnnotations,
260
+ egressMode: config.egressMode,
261
+ egressAllowFqdns: [...adapterDefaults.allowFqdns, ...config.egressAllowFqdns],
262
+ egressAllowCidrs: config.egressAllowCidrs,
263
+ resourceQuota: DEFAULT_RESOURCE_QUOTA,
264
+ });
265
+ const jobName = `pc-${newRunUlidDns()}`;
266
+ const secretName = `${jobName}-env`;
267
+ // TODO: use params.runId as stand-in for agentId in labels; future
268
+ // versions will have a dedicated agentId on AcquireLeaseParams.
269
+ const labels = paperclipLabels({
270
+ runId: params.runId,
271
+ agentId: params.runId,
272
+ companyId: params.companyId,
273
+ adapterType: effectiveAdapterType,
274
+ });
275
+ const image = resolveImage({ imageOverride: null }, adapterDefaults, { imageAllowList: config.imageAllowList, imageRegistry: config.imageRegistry });
276
+ // Pick the orchestrator and build the appropriate manifest based on backend.
277
+ const isSandboxCrBackend = config.backend === "sandbox-cr";
278
+ const orchestrator = isSandboxCrBackend ? sandboxCrOrchestrator : jobOrchestrator;
279
+ const manifest = isSandboxCrBackend
280
+ ? buildSandboxCrManifest({
281
+ namespace,
282
+ sandboxName: jobName,
283
+ adapterType: effectiveAdapterType,
284
+ image,
285
+ envSecretName: secretName,
286
+ serviceAccountName: TENANT_SERVICE_ACCOUNT,
287
+ labels,
288
+ resources: config.defaultResources ?? {},
289
+ runtimeClassName: config.runtimeClassName,
290
+ imagePullSecrets: config.imagePullSecrets,
291
+ })
292
+ : buildJobManifest({
293
+ namespace,
294
+ jobName,
295
+ adapterType: effectiveAdapterType,
296
+ image,
297
+ envSecretName: secretName,
298
+ serviceAccountName: TENANT_SERVICE_ACCOUNT,
299
+ labels,
300
+ resources: config.defaultResources ?? {},
301
+ runtimeClassName: config.runtimeClassName,
302
+ activeDeadlineSec: config.podActivityDeadlineSec,
303
+ ttlSecondsAfterFinished: config.jobTtlSecondsAfterFinished,
304
+ imagePullSecrets: config.imagePullSecrets,
305
+ });
306
+ const { uid: ownerUid } = await orchestrator.claim(clients, namespace, manifest);
307
+ const scopedNetworkEgress = parseScopedNetworkEgressGrant(params.executionWorkspaceSettings);
308
+ const scopedNetworkPolicyName = await createScopedNetworkEgressPolicyOrReleaseWorkload({
309
+ clients,
310
+ namespace,
311
+ mode: config.egressMode,
312
+ runId: params.runId,
313
+ workloadName: jobName,
314
+ ownerReference: {
315
+ apiVersion: isSandboxCrBackend ? "agents.x-k8s.io/v1alpha1" : "batch/v1",
316
+ kind: isSandboxCrBackend ? "Sandbox" : "Job",
317
+ name: jobName,
318
+ uid: ownerUid,
319
+ controller: false,
320
+ blockOwnerDeletion: false,
321
+ },
322
+ grant: scopedNetworkEgress,
323
+ }, () => orchestrator.release(clients, namespace, jobName));
324
+ // defaultEnv (non-secret base, e.g. the inference base URL) is layered first;
325
+ // the process-env secrets named by envKeys override it.
326
+ const adapterEnv = buildAdapterEnv(adapterDefaults);
327
+ adapterEnv.PAPERCLIP_NETWORK_EGRESS_POLICY = "kubernetes-default-deny";
328
+ adapterEnv.PAPERCLIP_NETWORK_EGRESS_GRANT_PATH = NETWORK_EGRESS_GRANT_PATH;
329
+ adapterEnv.PAPERCLIP_NETWORK_EGRESS_ALLOW_FQDNS = scopedNetworkEgress.allowFqdns.join(",");
330
+ adapterEnv.PAPERCLIP_NETWORK_EGRESS_ALLOW_CIDRS = scopedNetworkEgress.allowCidrs.join(",");
331
+ const bootstrapToken = generateBootstrapToken();
332
+ // Secret ownerRef: for job backend, the Job owns the Secret (cascade delete).
333
+ // For sandbox-cr backend, the Sandbox CR owns the Secret.
334
+ // NOTE: For sandbox-cr, if the Secret outlives the Sandbox due to a cluster
335
+ // quirk, the release() call will still clean it up via namespace GC or
336
+ // explicit delete in a future iteration.
337
+ await createPerRunSecret(clients, {
338
+ namespace,
339
+ secretName,
340
+ runId: params.runId,
341
+ ownerKind: isSandboxCrBackend ? "Sandbox" : "Job",
342
+ ownerApiVersion: isSandboxCrBackend ? "agents.x-k8s.io/v1alpha1" : "batch/v1",
343
+ ownerName: jobName,
344
+ ownerUid,
345
+ bootstrapToken,
346
+ adapterEnv,
347
+ });
348
+ const podName = await orchestrator.findPod(clients, namespace, jobName);
349
+ const leaseMetadata = {
350
+ namespace,
351
+ jobName,
352
+ podName,
353
+ secretName,
354
+ phase: "Pending",
355
+ backend: config.backend,
356
+ scopedNetworkPolicyName,
357
+ scopedNetworkEgress,
358
+ // Native file sync streams over a pod exec; only the sandbox-cr backend
359
+ // exposes one. Flag the job backend so the server keeps the base64 fallback
360
+ // rather than routing its sync to a hook that would reject immediately.
361
+ nativeFileSyncUnsupported: config.backend !== "sandbox-cr",
362
+ };
363
+ return {
364
+ providerLeaseId: jobName,
365
+ metadata: leaseMetadata,
366
+ };
367
+ },
368
+ async onEnvironmentResumeLease(params) {
369
+ const config = kubernetesProviderConfigSchema.parse(params.config);
370
+ const namespace = typeof params.leaseMetadata?.namespace === "string"
371
+ ? params.leaseMetadata.namespace
372
+ : deriveTenantNamespace(config, params.companyId);
373
+ const leaseBackend = typeof params.leaseMetadata?.backend === "string"
374
+ ? params.leaseMetadata.backend
375
+ : config.backend;
376
+ // acquireLease names the per-run Secret `${jobName}-env` and uses jobName
377
+ // as the providerLeaseId, so the suffix fallback reconstructs it exactly.
378
+ const secretName = typeof params.leaseMetadata?.secretName === "string"
379
+ ? params.leaseMetadata.secretName
380
+ : `${params.providerLeaseId}-env`;
381
+ const kc = createKubeConfig({
382
+ inCluster: config.inCluster,
383
+ kubeconfig: config.kubeconfig,
384
+ });
385
+ const clients = makeKubeClients(kc);
386
+ const check = await checkLeaseResumable(clients, {
387
+ namespace,
388
+ name: params.providerLeaseId,
389
+ backend: leaseBackend,
390
+ readyTimeoutMs: RESUME_READY_TIMEOUT_MS,
391
+ pollMs: RESUME_READY_POLL_MS,
392
+ });
393
+ if (!check.resumable) {
394
+ // Kubernetes pods are NOT restartable the way Daytona sandboxes are: a
395
+ // stopped Daytona sandbox can be started again by ID, but a k8s pod that
396
+ // is gone or terminally failed can never be revived in place. Gone = not
397
+ // resumable, by design. Returning providerLeaseId: null tells the server
398
+ // the lease expired so it falls back to a fresh acquireLease.
399
+ return {
400
+ providerLeaseId: null,
401
+ metadata: { expired: true, reason: check.reason },
402
+ };
403
+ }
404
+ // A resumed lease starts with clean per-lease state: drop any stale upload
405
+ // interceptor buffers a previous run on this lease may have left behind.
406
+ uploadInterceptorsByLease.delete(params.providerLeaseId);
407
+ if (leaseBackend === "sandbox-cr") {
408
+ // We just observed the Sandbox pod Ready, so the first exec on the
409
+ // resumed lease can skip its readiness poll.
410
+ readySandboxesByLease.add(params.providerLeaseId);
411
+ }
412
+ const leaseMetadata = {
413
+ namespace,
414
+ jobName: params.providerLeaseId,
415
+ podName: check.podName,
416
+ secretName,
417
+ phase: check.phase,
418
+ backend: leaseBackend,
419
+ scopedNetworkPolicyName: typeof params.leaseMetadata?.scopedNetworkPolicyName === "string"
420
+ ? params.leaseMetadata.scopedNetworkPolicyName
421
+ : null,
422
+ scopedNetworkEgress: parseScopedNetworkEgressGrant({
423
+ networkEgress: params.leaseMetadata?.scopedNetworkEgress,
424
+ }),
425
+ // See acquireLease: only the sandbox-cr backend has a pod-exec channel for
426
+ // native sync, so a resumed job lease must keep the base64 fallback.
427
+ nativeFileSyncUnsupported: leaseBackend !== "sandbox-cr",
428
+ };
429
+ return {
430
+ providerLeaseId: params.providerLeaseId,
431
+ metadata: {
432
+ ...leaseMetadata,
433
+ resumedLease: true,
434
+ },
435
+ };
436
+ },
437
+ async onEnvironmentRealizeWorkspace(params) {
438
+ // The agent pod already has /workspace mounted as an emptyDir at pod
439
+ // scheduling time (see pod-spec-builder). Nothing to provision here —
440
+ // we just hand back the cwd. Honor a caller-supplied remotePath if set.
441
+ const cwd = params.workspace.remotePath && params.workspace.remotePath.trim().length > 0
442
+ ? params.workspace.remotePath.trim()
443
+ : "/workspace";
444
+ return {
445
+ cwd,
446
+ metadata: {
447
+ provider: "kubernetes",
448
+ remoteCwd: cwd,
449
+ },
450
+ };
451
+ },
452
+ async onEnvironmentReleaseLease(params) {
453
+ if (!params.providerLeaseId)
454
+ return;
455
+ const config = kubernetesProviderConfigSchema.parse(params.config);
456
+ const namespace = typeof params.leaseMetadata?.namespace === "string"
457
+ ? params.leaseMetadata.namespace
458
+ : deriveTenantNamespace(config, params.companyId);
459
+ const kc = createKubeConfig({
460
+ inCluster: config.inCluster,
461
+ kubeconfig: config.kubeconfig,
462
+ });
463
+ const clients = makeKubeClients(kc);
464
+ const leaseBackend = typeof params.leaseMetadata?.backend === "string"
465
+ ? params.leaseMetadata.backend
466
+ : config.backend;
467
+ const releaseOrchestrator = leaseBackend === "sandbox-cr" ? sandboxCrOrchestrator : jobOrchestrator;
468
+ // Drop the FastUploadInterceptor associated with THIS lease (only).
469
+ // Each lease has its own interceptor instance via uploadInterceptorsByLease,
470
+ // so unrelated concurrent leases keep their in-flight buffers intact.
471
+ uploadInterceptorsByLease.delete(params.providerLeaseId);
472
+ readySandboxesByLease.delete(params.providerLeaseId);
473
+ try {
474
+ await releaseOrchestrator.release(clients, namespace, params.providerLeaseId);
475
+ }
476
+ catch (err) {
477
+ // If the resource is already gone (404), that's fine.
478
+ const code = err.code
479
+ ?? err.statusCode;
480
+ if (code !== 404)
481
+ throw err;
482
+ }
483
+ },
484
+ async onEnvironmentDestroyLease(params) {
485
+ if (!params.providerLeaseId)
486
+ return;
487
+ const config = kubernetesProviderConfigSchema.parse(params.config);
488
+ const namespace = typeof params.leaseMetadata?.namespace === "string"
489
+ ? params.leaseMetadata.namespace
490
+ : deriveTenantNamespace(config, params.companyId);
491
+ const leaseBackend = typeof params.leaseMetadata?.backend === "string"
492
+ ? params.leaseMetadata.backend
493
+ : config.backend;
494
+ const secretName = typeof params.leaseMetadata?.secretName === "string"
495
+ ? params.leaseMetadata.secretName
496
+ : `${params.providerLeaseId}-env`;
497
+ const podName = typeof params.leaseMetadata?.podName === "string" &&
498
+ params.leaseMetadata.podName.length > 0
499
+ ? params.leaseMetadata.podName
500
+ : null;
501
+ // Clear per-lease in-memory state up front, regardless of what the
502
+ // cluster says — the lease is dead either way.
503
+ uploadInterceptorsByLease.delete(params.providerLeaseId);
504
+ readySandboxesByLease.delete(params.providerLeaseId);
505
+ const kc = createKubeConfig({
506
+ inCluster: config.inCluster,
507
+ kubeconfig: config.kubeconfig,
508
+ });
509
+ const clients = makeKubeClients(kc);
510
+ // Forcibly delete everything acquireLease created (Sandbox CR / Job, pod,
511
+ // per-run Secret). 404s are success — destroy must be idempotent.
512
+ await destroyLeaseResources(clients, {
513
+ namespace,
514
+ name: params.providerLeaseId,
515
+ backend: leaseBackend,
516
+ podName,
517
+ secretName,
518
+ });
519
+ },
520
+ async onEnvironmentExecute(params) {
521
+ const { lease, timeoutMs } = params;
522
+ if (!lease.providerLeaseId) {
523
+ return {
524
+ exitCode: 1,
525
+ timedOut: false,
526
+ stdout: "",
527
+ stderr: "No provider lease ID available for execution.",
528
+ };
529
+ }
530
+ const config = kubernetesProviderConfigSchema.parse(params.config);
531
+ const scopedNetworkEgress = parseScopedNetworkEgressGrant({
532
+ networkEgress: lease.metadata?.scopedNetworkEgress,
533
+ });
534
+ const namespace = typeof lease.metadata?.namespace === "string"
535
+ ? lease.metadata.namespace
536
+ : deriveTenantNamespace(config, params.companyId);
537
+ // Determine which backend this lease was created with.
538
+ const leaseBackend = typeof lease.metadata?.backend === "string"
539
+ ? lease.metadata.backend
540
+ : config.backend;
541
+ const kc = createKubeConfig({
542
+ inCluster: config.inCluster,
543
+ kubeconfig: config.kubeconfig,
544
+ });
545
+ const clients = makeKubeClients(kc);
546
+ const effectiveTimeoutMs = typeof timeoutMs === "number" && timeoutMs > 0
547
+ ? timeoutMs
548
+ : config.podActivityDeadlineSec * 1000;
549
+ if (leaseBackend === "sandbox-cr") {
550
+ // ── Sandbox-CR backend ──────────────────────────────────────────────────
551
+ // 1. Ensure the Sandbox pod is Ready (wait only on first exec for this lease).
552
+ // 2. Exec the command into the running pod.
553
+ // 3. Return exec result directly (no log scraping needed).
554
+ let podName = typeof lease.metadata?.podName === "string" && lease.metadata.podName
555
+ ? lease.metadata.podName
556
+ : null;
557
+ // Skip the readiness poll if we've already observed this Sandbox CR
558
+ // reaching Ready during this worker's lifetime. See readySandboxesByLease
559
+ // declaration for rationale.
560
+ const podAlreadyKnownReady = readySandboxesByLease.has(lease.providerLeaseId);
561
+ // The caller's timeout is a budget for the WHOLE execute call: readiness
562
+ // wait + exec must share it, or the first exec on a fresh lease could
563
+ // block for up to twice the requested timeout.
564
+ const executeStartedAt = Date.now();
565
+ if (!podAlreadyKnownReady) {
566
+ try {
567
+ await sandboxCrOrchestrator.waitForCompletion(clients, namespace, lease.providerLeaseId, { timeoutMs: effectiveTimeoutMs, pollMs: 2000 });
568
+ readySandboxesByLease.add(lease.providerLeaseId);
569
+ }
570
+ catch (err) {
571
+ if (err instanceof SandboxCrTimeoutError) {
572
+ return {
573
+ exitCode: null,
574
+ timedOut: true,
575
+ stdout: "",
576
+ stderr: `Sandbox pod did not become Ready within ${effectiveTimeoutMs}ms`,
577
+ metadata: {
578
+ provider: "kubernetes",
579
+ backend: "sandbox-cr",
580
+ namespace,
581
+ sandboxName: lease.providerLeaseId,
582
+ },
583
+ };
584
+ }
585
+ throw err;
586
+ }
587
+ }
588
+ // Resolve pod name (may now be populated in Sandbox status).
589
+ if (!podName) {
590
+ podName = await sandboxCrOrchestrator.findPod(clients, namespace, lease.providerLeaseId);
591
+ }
592
+ if (!podName) {
593
+ return {
594
+ exitCode: 1,
595
+ timedOut: false,
596
+ stdout: "",
597
+ stderr: "Sandbox pod is Ready but podName could not be resolved.",
598
+ metadata: {
599
+ provider: "kubernetes",
600
+ backend: "sandbox-cr",
601
+ namespace,
602
+ sandboxName: lease.providerLeaseId,
603
+ },
604
+ };
605
+ }
606
+ // Build the command to exec. The adapter passes shell invocations as
607
+ // `command: "sh", args: ["-c", "<script>"]` — must combine both, NOT
608
+ // drop args. If only command is present (no args), wrap in a login shell.
609
+ const command = typeof params.command === "string" ? params.command.trim() : "";
610
+ const args = Array.isArray(params.args) ? params.args : [];
611
+ // Fast-upload interceptor: short-circuit the chunked-shell file transfer
612
+ // protocol (adapter-utils writeFile) so an N-chunk upload becomes 1 exec
613
+ // instead of N+2. Falls back transparently when patterns don't match.
614
+ // See upload-interceptor.ts.
615
+ const shellScript = command === "sh" && args[0] === "-c" && typeof args[1] === "string"
616
+ ? args[1]
617
+ : null;
618
+ if (shellScript) {
619
+ const decision = getOrCreateUploadInterceptor(lease.providerLeaseId).decide(shellScript);
620
+ if (decision.action === "ack") {
621
+ return {
622
+ exitCode: 0,
623
+ timedOut: false,
624
+ stdout: "",
625
+ stderr: "",
626
+ metadata: {
627
+ provider: "kubernetes",
628
+ backend: "sandbox-cr",
629
+ namespace,
630
+ sandboxName: lease.providerLeaseId,
631
+ podName,
632
+ fastUpload: "ack",
633
+ },
634
+ };
635
+ }
636
+ if (decision.action === "flush") {
637
+ // Single exec: `head -c <N> | base64 -d > '<TARGET>'` with stdin =
638
+ // base64 ASCII. `head -c` reads EXACTLY N bytes and exits, so we
639
+ // don't depend on WebSocket-driven EOF detection on stdin (which is
640
+ // racy against the `base64 -d` exit timing in @kubernetes/client-node
641
+ // v0.21.0 — see pod-exec.ts). All bytes are sent through the
642
+ // WebSocket data channel; size is unbounded by ARG_MAX.
643
+ const base64Body = decision.flush.payload.toString("base64");
644
+ const dir = decision.flush.targetPath.substring(0, decision.flush.targetPath.lastIndexOf("/"));
645
+ const script = `mkdir -p '${dir}' && ` +
646
+ `head -c ${base64Body.length} | base64 -d > '${decision.flush.targetPath}'`;
647
+ // The flush shares the caller's single execute budget (same contract
648
+ // as the normal exec path below) and surfaces watchdog/WebSocket
649
+ // failures as a timed-out result instead of an uncaught throw.
650
+ const flushTimeoutMs = Math.max(5_000, effectiveTimeoutMs - (Date.now() - executeStartedAt));
651
+ let flushResult;
652
+ try {
653
+ flushResult = await execInPod(kc, namespace, podName, "agent", ["/bin/sh", "-c", script], base64Body, flushTimeoutMs);
654
+ }
655
+ catch (err) {
656
+ return {
657
+ exitCode: null,
658
+ timedOut: true,
659
+ stdout: "",
660
+ stderr: `fast-upload flush failed: ${err instanceof Error ? err.message : String(err)}`,
661
+ metadata: {
662
+ provider: "kubernetes",
663
+ backend: "sandbox-cr",
664
+ namespace,
665
+ sandboxName: lease.providerLeaseId,
666
+ podName,
667
+ fastUpload: "flush",
668
+ },
669
+ };
670
+ }
671
+ return {
672
+ exitCode: flushResult.exitCode,
673
+ timedOut: false,
674
+ stdout: flushResult.stdout,
675
+ stderr: flushResult.stderr,
676
+ metadata: {
677
+ provider: "kubernetes",
678
+ backend: "sandbox-cr",
679
+ namespace,
680
+ sandboxName: lease.providerLeaseId,
681
+ podName,
682
+ fastUpload: "flush",
683
+ uploadedBytes: decision.flush.payload.length,
684
+ },
685
+ };
686
+ }
687
+ // decision.action === "passthrough" — fall through to normal exec
688
+ }
689
+ const baseExecCommand = command.length > 0 && args.length > 0
690
+ ? [command, ...args]
691
+ : command.length > 0
692
+ ? ["/bin/sh", "-lc", command]
693
+ : ["/bin/sh", "-l"];
694
+ // Apply the caller-provided run env (params.env) to the in-pod process. Without
695
+ // this the adapter's runtime env (e.g. XDG_CONFIG_HOME pointing at the shipped
696
+ // OpenCode config, plus helper settings like small_model/provider routing) never
697
+ // reaches the harness, which falls back to its in-image HOME config -> wrong or
698
+ // partial behaviour.
699
+ const execCommand = wrapCommandWithEnv(baseExecCommand, params.env);
700
+ // Remaining share of the caller's budget after the readiness wait (floor
701
+ // of 5s so an exec attempt is still made when readiness consumed most of
702
+ // it; the watchdog then bounds it tightly).
703
+ const remainingTimeoutMs = Math.max(5_000, effectiveTimeoutMs - (Date.now() - executeStartedAt));
704
+ let execResult;
705
+ try {
706
+ execResult = await execInPod(kc, namespace, podName, "agent", execCommand, typeof params.stdin === "string" ? params.stdin : undefined, remainingTimeoutMs);
707
+ }
708
+ catch (err) {
709
+ // Watchdog-fired or WebSocket-setup error. Surface as a timeout so
710
+ // the caller can retry instead of hanging forever.
711
+ return {
712
+ exitCode: null,
713
+ timedOut: true,
714
+ stdout: "",
715
+ stderr: appendNetworkEgressDenyHint(err instanceof Error ? err.message : String(err), scopedNetworkEgress),
716
+ metadata: {
717
+ provider: "kubernetes",
718
+ backend: "sandbox-cr",
719
+ namespace,
720
+ sandboxName: lease.providerLeaseId,
721
+ podName,
722
+ },
723
+ };
724
+ }
725
+ return {
726
+ exitCode: execResult.exitCode,
727
+ timedOut: false,
728
+ stdout: execResult.stdout,
729
+ stderr: appendNetworkEgressDenyHint(execResult.stderr, scopedNetworkEgress),
730
+ metadata: {
731
+ provider: "kubernetes",
732
+ backend: "sandbox-cr",
733
+ namespace,
734
+ sandboxName: lease.providerLeaseId,
735
+ podName,
736
+ },
737
+ };
738
+ }
739
+ else {
740
+ // ── Job backend (legacy / stable fallback) ──────────────────────────────
741
+ // The container entrypoint is baked into the Job spec (Tini + paperclip-agent-shim).
742
+ // We do NOT re-exec command/args — instead we wait for the Job to finish
743
+ // and collect its logs.
744
+ //
745
+ // params.command / params.args / params.stdin are intentionally ignored.
746
+ let status;
747
+ let timedOut = false;
748
+ try {
749
+ status = await jobOrchestrator.waitForCompletion(clients, namespace, lease.providerLeaseId, { timeoutMs: effectiveTimeoutMs, pollMs: 2000 });
750
+ }
751
+ catch (err) {
752
+ if (err instanceof JobTimeoutError) {
753
+ timedOut = true;
754
+ status = null;
755
+ }
756
+ else {
757
+ throw err;
758
+ }
759
+ }
760
+ // Collect logs from the pod.
761
+ const podName = typeof lease.metadata?.podName === "string"
762
+ ? lease.metadata.podName
763
+ : await jobOrchestrator.findPod(clients, namespace, lease.providerLeaseId);
764
+ const stdoutChunks = [];
765
+ const stderrChunks = [];
766
+ if (podName) {
767
+ await jobOrchestrator.streamLogs(clients, namespace, podName, async (stream, text) => {
768
+ if (stream === "stdout")
769
+ stdoutChunks.push(text);
770
+ else
771
+ stderrChunks.push(text);
772
+ });
773
+ }
774
+ return {
775
+ exitCode: timedOut ? null : status?.phase === "Succeeded" ? 0 : 1,
776
+ timedOut,
777
+ stdout: stdoutChunks.join(""),
778
+ stderr: appendNetworkEgressDenyHint(stderrChunks.join(""), scopedNetworkEgress),
779
+ metadata: {
780
+ provider: "kubernetes",
781
+ backend: "job",
782
+ namespace,
783
+ jobName: lease.providerLeaseId,
784
+ podName: podName ?? null,
785
+ phase: status?.phase ?? null,
786
+ },
787
+ };
788
+ }
789
+ },
790
+ // Opt-in native inbound transfer. Defining this hook (with onEnvironmentSyncOut)
791
+ // makes the worker advertise `environmentSyncIn`/`environmentSyncOut`, so the
792
+ // host runner routes workspace/asset transfers through a single pod exec per
793
+ // operation (host tar streamed over the exec stdin → in-pod `head -c <N> | tar
794
+ // -x` → stage-then-atomic-`mv -f`) instead of the base64-over-exec chunk loop.
795
+ // Only the sandbox-cr backend is supported; the job backend carries no file
796
+ // path. Providers that do not define these keep the byte-identical fallback.
797
+ async onEnvironmentSyncIn(params) {
798
+ const remoteDir = resolveSyncRemoteDir(params.lease);
799
+ const { exec, timeoutMs } = await resolveSyncPodExec(params);
800
+ return await performSyncIn({
801
+ exec,
802
+ operations: params.operations,
803
+ remoteDir,
804
+ timeoutMs,
805
+ });
806
+ },
807
+ // Opt-in native outbound transfer. See onEnvironmentSyncIn.
808
+ async onEnvironmentSyncOut(params) {
809
+ const remoteDir = resolveSyncRemoteDir(params.lease);
810
+ const { exec, timeoutMs } = await resolveSyncPodExec(params);
811
+ return await performSyncOut({
812
+ exec,
813
+ operations: params.operations,
814
+ remoteDir,
815
+ timeoutMs,
816
+ });
817
+ },
818
+ });
819
+ export default plugin;
820
+ //# sourceMappingURL=plugin.js.map