@evolvingmachines/sdk 0.0.50 → 0.0.52-launch-round-1.20260803.37a56fe

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.cts CHANGED
@@ -1,9 +1,11 @@
1
1
  import { EventEmitter } from 'events';
2
2
  import * as zod from 'zod';
3
- import { z } from 'zod';
3
+ import { ZodType, z } from 'zod';
4
4
  export { E2BConfig, E2BProvider, createE2BProvider } from '@evolvingmachines/e2b';
5
5
  export { DaytonaConfig, DaytonaProvider, createDaytonaProvider } from '@evolvingmachines/daytona';
6
6
  export { ModalConfig, ModalProvider, createModalProvider } from '@evolvingmachines/modal';
7
+ import { H as HostedClientConfig, A as AgentsClient, a as AuthClient, D as DatasetsClient, J as JobsClient, T as TrialsClient, C as CapabilityDocument, b as HostedErrorCode } from './types-B6iGD_du.cjs';
8
+ export { r as ActiveDataset, y as AgentArm, z as AgentArmInput, f as AgentCapability, K as AgentDatasetStats, _ as AgentInfo, aq as AgentInput, aE as AgentList, aD as AgentPage, $ as AgentResult, as as AgentSource, at as AgentSourceInput, ar as AgentUpsertInput, l as ApiKey, a3 as AttemptPhase, k as AuthStatus, g as Awaitable, ab as CompareCell, ac as CompareCoverage, aa as CompareJobAggregate, ae as CompareResponse, ad as CompareTaskRow, q as Dataset, ai as DatasetImport, an as DatasetImportFailure, h as DatasetImportList, j as DatasetImportPage, am as DatasetImportStatus, aC as DatasetList, aB as DatasetPage, aj as DatasetPatch, u as DatasetRef, v as DatasetSelector, al as DatasetSource, s as DatasetVersion, t as DatasetVersionState, aP as DownloadDatasetOptions, aO as DownloadJobOptions, E as EVAL_SANDBOX_PROVIDERS, ap as EvalAgent, Z as EvalModelInfo, O as EvalSandboxProvider, a2 as EvalStepResult, a1 as ExceptionInfo, aK as GetDatasetOptions, c as HOSTED_ERROR_CODES, ao as ImportWarning, F as Job, B as JobCreate, a7 as JobEvent, G as JobFailure, ay as JobList, ax as JobPage, I as JobStats, N as JobStatus, m as JobTaskRollup, o as JobTaskRollupList, n as JobTaskRollupPage, a8 as JobWatch, aJ as ListAgentsOptions, aI as ListDatasetsOptions, L as ListImportsOptions, p as ListJobTasksOptions, aG as ListJobsOptions, aH as ListTrialsOptions, M as ManagedProviderCapability, av as Page, aw as PageOptions, P as ProviderCapability, ak as PublishDatasetInput, ah as RegradeRequest, ag as ResumeRequest, af as SourceJob, au as SpendSource, aF as StartJobOptions, S as StatusVocabulary, a4 as StopResponse, d as TRIAL_ARTIFACT_STREAMS, e as TRIAL_STATUSES, w as Task, x as TaskProviderVerdict, Y as TimingInfo, a5 as TraceEvent, a6 as TraceEventPage, aL as TraceOptions, Q as Trial, R as TrialArtifactStream, W as TrialCounts, aA as TrialList, az as TrialPage, V as TrialStatus, X as TrialStatusTally, U as UpstreamStatus, a9 as VerifierEnvironmentMode, a0 as VerifierResult, aN as WatchImportOptions, aM as WatchJobOptions, i as isHostedErrorCode } from './types-B6iGD_du.cjs';
7
9
 
8
10
  /**
9
11
  * ACP-inspired output types for unified agent event streaming.
@@ -141,7 +143,40 @@ interface PlanEntry {
141
143
  * All possible session update types.
142
144
  * Discriminated union on `sessionUpdate` field.
143
145
  */
144
- type SessionUpdate = AgentMessageChunk | AgentThoughtChunk | UserMessageChunk | ToolCall | ToolCallUpdate | Plan;
146
+ type SessionUpdate = AgentMessageChunk | AgentThoughtChunk | UserMessageChunk | ToolCall | ToolCallUpdate | Plan | AgentError;
147
+ /**
148
+ * A failure the HARNESS itself reported — not model output, not work.
149
+ *
150
+ * WHY THIS IS ITS OWN VARIANT AND NOT AN agent_message_chunk. Harnesses stream
151
+ * their failures on the same channel as their output: codex writes
152
+ * {"type":"error"} and {"type":"turn.failed"} to stdout as JSONL while its
153
+ * stderr says only "Reading prompt from stdin...". Dropping those left a run
154
+ * that could not reach the model looking identical to a run that produced
155
+ * nothing at all, which cost a full night of blind diagnosis. Folding them into
156
+ * agent_message_chunk would be worse than dropping them: a consumer counting
157
+ * "did the agent do any work" would count the error as work.
158
+ *
159
+ * So the transcript records the failure, and the discriminant says plainly that
160
+ * it is a failure. Anything deciding whether a harness RAN must exclude this
161
+ * variant — see isAgentWorkUpdate() below, and the eval runner's
162
+ * harnessNeverRan law, which must keep firing for an error-only run so an
163
+ * infrastructure failure is never scored as a zero.
164
+ */
165
+ interface AgentError {
166
+ sessionUpdate: "error";
167
+ /** The harness's own message, verbatim. */
168
+ message: string;
169
+ /** True when the harness treated it as terminal for the turn. */
170
+ fatal: boolean;
171
+ }
172
+ /**
173
+ * Is this update evidence the harness did WORK, as opposed to reporting a
174
+ * failure? The one predicate every "did it run" check should use, so the answer
175
+ * cannot drift between callers.
176
+ */
177
+ declare function isAgentWorkUpdate(update: {
178
+ sessionUpdate?: unknown;
179
+ } | null | undefined): boolean;
145
180
  /**
146
181
  * Streaming text/image from agent.
147
182
  * May arrive in multiple chunks - concatenate text.
@@ -285,6 +320,301 @@ interface OutputEvent {
285
320
  update: SessionUpdate;
286
321
  }
287
322
 
323
+ /**
324
+ * Agent Registry
325
+ *
326
+ * Single source of truth for agent-specific behavior.
327
+ * All differences between agents are data, not code.
328
+ */
329
+
330
+ /**
331
+ * What a CLI does with a reasoning-effort input:
332
+ * 'level' the value reaches the CLI as a graded level
333
+ * 'binary' thinking on/off only — only BINARY_EFFORT_VALUES are honest inputs
334
+ * 'none' the CLI takes no effort input at all
335
+ */
336
+ type EffortSupport = "level" | "binary" | "none";
337
+ /**
338
+ * The effort vocabulary a graded ('level') harness accepts, in ascending order
339
+ * of thinking. This is the ADVERTISED list: the ReasoningEffort union in
340
+ * types.ts additionally accepts the legacy spelling "no-thinking" (same
341
+ * behaviour as "off"), which is deliberately not advertised anywhere.
342
+ */
343
+ declare const REASONING_EFFORTS: readonly ["off", "minimal", "low", "medium", "high", "xhigh", "max", "thinking"];
344
+ /**
345
+ * The subset a 'binary' harness can honestly represent. Four SPELLINGS, two
346
+ * BEHAVIOURS: 'off' and 'minimal' disable thinking, 'medium' and 'thinking'
347
+ * enable it (isThinkingEnabled draws the same line). The graded levels are
348
+ * excluded because a binary CLI cannot express a gradation — accepting 'high'
349
+ * would record a claim the CLI never received.
350
+ */
351
+ declare const BINARY_EFFORT_VALUES: readonly ["off", "minimal", "medium", "thinking"];
352
+ /**
353
+ * The effort a run takes when it names none, for harnesses that take one at
354
+ * all. Owned HERE because managed evals and managed agents must advertise the
355
+ * same defaults: the hosted-evals lane resolves an omitted effort to this value
356
+ * at job creation and publishes it on GET /api/meta via the generated
357
+ * harness-capabilities.json artifact.
358
+ */
359
+ declare const DEFAULT_REASONING_EFFORT: ReasoningEffort;
360
+ /**
361
+ * The effort vocabulary one harness accepts and the value an unnamed effort
362
+ * takes there — pure data derivation from its effortSupport. The
363
+ * harness-capabilities artifact generator and picker UIs share this.
364
+ */
365
+ declare function harnessEffortVocabulary(support: EffortSupport): {
366
+ efforts: readonly ReasoningEffort[];
367
+ defaultEffort: ReasoningEffort | null;
368
+ };
369
+ /** Model configuration */
370
+ interface ModelInfo {
371
+ /** Model alias (short name used with --model) */
372
+ alias: string;
373
+ /** Full model ID */
374
+ modelId: string;
375
+ /** What this model is best for */
376
+ description: string;
377
+ /** Per-model context ceiling where it differs from the harness default
378
+ * (e.g. Kimi K3's 1M window vs the K2-era 262144). Consumers fall back to
379
+ * the harness-level value when absent. */
380
+ maxContextSize?: number;
381
+ }
382
+ /** MCP configuration for an agent */
383
+ interface McpConfigInfo {
384
+ /** Settings directory (e.g., "~/.claude") */
385
+ settingsDir: string;
386
+ /** Config filename (e.g., "settings.json" or "config.toml") */
387
+ filename: string;
388
+ /** Config format */
389
+ format: "json" | "toml";
390
+ /** Whether to use workingDir for project-level config (Claude only) */
391
+ projectConfig?: boolean;
392
+ }
393
+ /** Options for building agent commands */
394
+ interface BuildCommandOptions {
395
+ prompt: string;
396
+ model: string;
397
+ isResume: boolean;
398
+ sessionId?: string;
399
+ reasoningEffort?: string;
400
+ isDirectMode?: boolean;
401
+ /**
402
+ * External gateway mode (caller-minted credential + base URL). Direct-mode
403
+ * env injection applies, but CLIs that route via generated config (OpenCode
404
+ * inline config, Droid settings file) must use their GATEWAY command shape
405
+ * pointed at the caller's gateway — with the model passed VERBATIM (route
406
+ * names belong to the caller's gateway, never to Evolve's alias maps).
407
+ */
408
+ isExternalGateway?: boolean;
409
+ /** Skills enabled for this run */
410
+ skills?: string[];
411
+ /** Sandbox home directory (default: "/home/user") */
412
+ homeDir?: string;
413
+ }
414
+ interface AgentRegistryEntry {
415
+ /** Sandbox image/template identifier (provider maps to its own concept) */
416
+ image: string;
417
+ /**
418
+ * What this CLI does with a reasoning-effort input (see EffortSupport).
419
+ * Advertised DATA beside the buildCommand BEHAVIOUR: the generated
420
+ * harness-capabilities.json artifact reads this so the hosted-evals lane
421
+ * advertises exactly the vocabulary the local SDK drives.
422
+ */
423
+ effortSupport: EffortSupport;
424
+ /** Environment variable name for API key */
425
+ apiKeyEnv: string;
426
+ /** Environment variable name for OAuth (file path or token depending on agent) */
427
+ oauthEnv?: string;
428
+ /** OAuth credentials filename (e.g., "auth.json" for Codex, "oauth_creds.json" for Gemini) */
429
+ oauthFileName?: string;
430
+ /** Environment variable to set when OAuth is active (e.g., GOOGLE_GENAI_USE_GCA=true for Gemini) */
431
+ oauthActivationEnv?: {
432
+ key: string;
433
+ value: string;
434
+ };
435
+ /** Environment variable name for base URL, if this CLI supports one */
436
+ baseUrlEnv?: string;
437
+ /** Default model alias */
438
+ defaultModel: string;
439
+ /**
440
+ * Reasoning effort Evolve pins when the caller omits `reasoningEffort`.
441
+ *
442
+ * Every run stamps this value explicitly on the wire (flag/env/config file)
443
+ * instead of relying on the vendor's silent default — managed-evals
444
+ * reproducibility requires the effort a run used to be recorded, not implied
445
+ * by whatever the CLI happened to default to that week. Where the vendor
446
+ * documents a default, the pin matches it; where none is documented, the pin
447
+ * is Evolve's choice (noted per entry). Absent only for harnesses with no
448
+ * effort control (Gemini).
449
+ */
450
+ defaultReasoningEffort?: ReasoningEffort;
451
+ /** Available models for this agent */
452
+ models: ModelInfo[];
453
+ /** System prompt filename (e.g., "CLAUDE.md") */
454
+ systemPromptFile: string;
455
+ /** MCP configuration */
456
+ mcpConfig: McpConfigInfo;
457
+ /** Build the CLI command for this agent */
458
+ buildCommand: (opts: BuildCommandOptions) => string;
459
+ /** Extra setup step (e.g., codex login) */
460
+ setupCommand?: string;
461
+ /** Gateway path prefix for CLIs that use a provider-native passthrough endpoint */
462
+ gatewayPath?: string;
463
+ /** Default base URL for direct mode (only needed if provider requires specific endpoint, e.g., Qwen → Dashscope) */
464
+ defaultBaseUrl?: string;
465
+ /** Available beta headers for this agent (for reference) */
466
+ availableBetas?: Record<string, string>;
467
+ /** Skills configuration for this agent */
468
+ skillsConfig: SkillsConfig;
469
+ /** Multi-provider env mapping: model prefix → keyEnv (for CLIs like OpenCode that resolve provider from model string) */
470
+ providerEnvMap?: Record<string, {
471
+ keyEnv: string;
472
+ }>;
473
+ /** Env var for inline config (e.g., OPENCODE_CONFIG_CONTENT) — used in gateway mode to set provider base URLs */
474
+ gatewayConfigEnv?: string;
475
+ /** Gateway-only model aliases for CLIs whose native model IDs differ from the Evolve gateway's route names */
476
+ gatewayModelAliases?: Record<string, string>;
477
+ /** Direct-mode model aliases for CLIs whose public model names differ from CLI-native model IDs */
478
+ directModelAliases?: Record<string, string>;
479
+ /** Do not set provider API key env in gateway mode (used when routing via generated settings instead) */
480
+ skipApiKeyEnvInGateway?: boolean;
481
+ /** Dedicated Droid settings file for Evolve gateway custom model routing */
482
+ droidGatewaySettings?: {
483
+ settingsPath: string;
484
+ displayName: string;
485
+ provider: "generic-chat-completion-api" | "openai" | "anthropic";
486
+ maxOutputTokens?: number;
487
+ };
488
+ /** Environment variable that CLI reads for custom outbound HTTP headers */
489
+ customHeadersEnv?: string;
490
+ /** Format for custom headers env var: "newline" (Claude) or "comma" (Gemini). Default: "newline" */
491
+ customHeadersFormat?: "newline" | "comma";
492
+ /**
493
+ * Per-env-var spend tracking for CLIs that support env_http_headers in config
494
+ * (e.g., Codex TOML). Maps Evolve gateway header names to env var names that the CLI
495
+ * reads at request time. Alternative to customHeadersEnv for agents without a
496
+ * single custom-headers env var.
497
+ */
498
+ spendTrackingEnvs?: {
499
+ /** Env var name for x-litellm-customer-id value */
500
+ sessionTagEnv: string;
501
+ /** Env var name for x-litellm-tags value */
502
+ runTagEnv: string;
503
+ };
504
+ /**
505
+ * Config-file-based spend tracking for CLIs that read custom headers from a
506
+ * JSON settings file (e.g., Qwen settings.json → model.generationConfig.customHeaders).
507
+ * The SDK writes headers to this file before each run.
508
+ * Source-verified: Qwen reads customHeaders from settings.json, not env vars.
509
+ */
510
+ spendTrackingJsonConfig?: {
511
+ /** JSON path to the customHeaders object (dot-separated) */
512
+ headersPath: string;
513
+ };
514
+ /**
515
+ * TOML provider-based spend tracking for CLIs that read custom_headers from a
516
+ * provider entry in config.toml (e.g., Kimi Code).
517
+ * The SDK writes a provider+model entry with custom_headers before each run.
518
+ * Source-verified: Kimi Code reads custom_headers from
519
+ * providers[name].custom_headers in ~/.kimi-code/config.toml.
520
+ */
521
+ spendTrackingTomlProvider?: {
522
+ /** Config file path (e.g., "~/.kimi-code/config.toml") */
523
+ configPath: string;
524
+ /** Provider name in config (e.g., "evolve-gateway") */
525
+ providerName: string;
526
+ /** Model entry name (e.g., "evolve-default") */
527
+ modelName: string;
528
+ /** Max context size for the model entry */
529
+ maxContextSize: number;
530
+ };
531
+ /** Additional directories to include in checkpoint tar (beyond mcpConfig.settingsDir).
532
+ * Used for agents like OpenCode that spread state across XDG directories. */
533
+ checkpointDirs?: string[];
534
+ /** Additional relative paths to exclude from checkpoint tar. */
535
+ checkpointExcludes?: string[];
536
+ }
537
+ /**
538
+ * Registry of all supported agents.
539
+ *
540
+ * Each agent defines a buildCommand function that constructs the CLI command.
541
+ * This is type-safe and handles conditional logic cleanly.
542
+ */
543
+ declare const AGENT_REGISTRY: Record<AgentType, AgentRegistryEntry>;
544
+ /**
545
+ * Get registry entry for an agent type
546
+ */
547
+ declare function getAgentConfig(agentType: AgentType): AgentRegistryEntry;
548
+ /**
549
+ * Check if an agent type is valid
550
+ */
551
+ declare function isValidAgentType(type: string): type is AgentType;
552
+ /**
553
+ * Expand path with ~ to the sandbox home directory (default: /home/user)
554
+ */
555
+ declare function expandPath(path: string, homeDir?: string): string;
556
+ /**
557
+ * Get MCP settings path for an agent
558
+ */
559
+ declare function getMcpSettingsPath(agentType: AgentType, homeDir?: string): string;
560
+ /**
561
+ * Get MCP settings directory for an agent
562
+ */
563
+ declare function getMcpSettingsDir(agentType: AgentType, homeDir?: string): string;
564
+
565
+ interface ManagedSecretRef {
566
+ name: string;
567
+ as?: string;
568
+ }
569
+ interface ManagedSecretMetadata {
570
+ id: string;
571
+ name: string;
572
+ allowedHosts: string[];
573
+ allowedPathPrefixes: string[];
574
+ allowedMethods: string[];
575
+ createdAt: string;
576
+ updatedAt: string;
577
+ lastUsedAt: string | null;
578
+ }
579
+ interface ManagedSecretsClientConfig {
580
+ apiKey?: string;
581
+ dashboardUrl?: string;
582
+ }
583
+ interface ManagedSecretsClient {
584
+ list(): Promise<ManagedSecretMetadata[]>;
585
+ }
586
+ declare function managedSecrets(config?: ManagedSecretsClientConfig): ManagedSecretsClient;
587
+
588
+ interface EvolveEvents {
589
+ stdout: (chunk: string) => void;
590
+ stderr: (chunk: string) => void;
591
+ content: (event: OutputEvent) => void;
592
+ lifecycle: (event: LifecycleEvent) => void;
593
+ }
594
+ interface EvolveConfig {
595
+ agent?: AgentConfig;
596
+ sandbox?: SandboxProvider;
597
+ sandboxCreateOptions?: SandboxCreateOptions;
598
+ workingDirectory?: string;
599
+ workspaceMode?: WorkspaceMode;
600
+ secrets?: Record<string, string>;
601
+ managedSecrets?: ManagedSecretRef[];
602
+ sandboxId?: string;
603
+ systemPrompt?: string;
604
+ context?: FileMap;
605
+ files?: FileMap;
606
+ mcpServers?: Record<string, McpServerConfig>;
607
+ browser?: BrowserConfig;
608
+ browserCredentials?: BrowserCredentialsConfig;
609
+ plugins?: AgentPluginConfig[];
610
+ skills?: SkillName[];
611
+ schema?: ZodType<unknown> | JsonSchema;
612
+ schemaOptions?: SchemaValidationOptions;
613
+ sessionTagPrefix?: string;
614
+ observability?: Record<string, unknown>;
615
+ integrations?: IntegrationsSetup;
616
+ storage?: StorageConfig;
617
+ }
288
618
  /** Result of a completed sandbox command */
289
619
  interface SandboxCommandResult {
290
620
  exitCode: number;
@@ -318,6 +648,13 @@ interface SandboxRunOptions {
318
648
  interface SandboxSpawnOptions extends SandboxRunOptions {
319
649
  stdin?: boolean;
320
650
  }
651
+ /** Provider-neutral outbound network policy applied when the sandbox boots. */
652
+ interface SandboxNetworkPolicy {
653
+ /** Allow all outbound traffic, or deny it except for allowedDestinations. */
654
+ outbound: "open" | "blocked";
655
+ /** Hostnames, IP addresses, or CIDR ranges that remain reachable when blocked. */
656
+ allowedDestinations?: string[];
657
+ }
321
658
  /** Options for creating a sandbox */
322
659
  interface SandboxCreateOptions {
323
660
  /** Sandbox image/template ID. Provider uses its default if not specified. */
@@ -325,7 +662,108 @@ interface SandboxCreateOptions {
325
662
  envs?: Record<string, string>;
326
663
  metadata?: Record<string, string>;
327
664
  timeoutMs?: number;
665
+ /**
666
+ * Terminate the sandbox after this long with nothing running in it. This is
667
+ * an INACTIVITY bound, not a lifetime: `timeoutMs` caps how long the box may
668
+ * live at all, this caps how long it may sit doing nothing — which is what
669
+ * reclaims a box whose client died between commands, long before the lifetime
670
+ * would.
671
+ *
672
+ * Providers must reject it if they cannot enforce it, never silently ignore
673
+ * it. Only modal has a true idle timer alongside an absolute lifetime; e2b has
674
+ * no idle concept, and daytona's only clock IS an inactivity one, which
675
+ * `timeoutMs` already drives there.
676
+ *
677
+ * WHAT COUNTS AS ACTIVITY IS THE PROVIDER'S DEFINITION, and it is narrower
678
+ * than "the client is doing something" — Modal counts a running exec, a stdin
679
+ * write, and an open tunnel connection, and says nothing about filesystem
680
+ * calls. Size this above the longest gap between commands the caller expects,
681
+ * not above the longest gap between API calls.
682
+ */
683
+ idleTimeoutMs?: number;
328
684
  workingDirectory?: string;
685
+ /**
686
+ * Per-sandbox compute sizing: cpu in cores, memory and disk in GiB.
687
+ * Providers must reject entries they cannot enforce at create time, never
688
+ * silently ignore them (modal sizes cpu/memory at create but cannot size
689
+ * disk; e2b sizes at template build only; daytona sizes at snapshot build,
690
+ * so an existing snapshot cannot be resized at create).
691
+ */
692
+ resources?: {
693
+ cpu?: number;
694
+ memory?: number;
695
+ disk?: number;
696
+ };
697
+ /** Providers must reject policies they cannot enforce; never silently ignore them. */
698
+ network?: SandboxNetworkPolicy;
699
+ /**
700
+ * Run all commands and file operations as this user.
701
+ * Providers must reject it if they cannot enforce it, never silently ignore it.
702
+ */
703
+ user?: string;
704
+ /**
705
+ * Home directory used for agent config paths inside the sandbox.
706
+ * Default: "/root" when user is "root", "/home/<user>" for other users,
707
+ * "/home/user" when no user is given.
708
+ */
709
+ homeDir?: string;
710
+ }
711
+ /** Options for listing sandboxes (capability: SandboxProvider.list). */
712
+ interface SandboxListOptions {
713
+ /**
714
+ * Provider-neutral states. Providers map them onto their own vocabulary and
715
+ * must not invent matches for a state they do not have (Modal has no paused
716
+ * state, so a filter excluding "running" matches nothing there).
717
+ */
718
+ state?: ("running" | "paused")[];
719
+ metadata?: Record<string, string>;
720
+ limit?: number;
721
+ }
722
+ /** File or directory entry (capability: SandboxFiles.list). */
723
+ interface FileInfo {
724
+ name: string;
725
+ path: string;
726
+ type: "file" | "dir";
727
+ }
728
+ /**
729
+ * A COMPLETE (or admittedly incomplete) enumeration of a provider's fleet.
730
+ *
731
+ * `complete` is the load-bearing field, not a nicety. The callers that need a
732
+ * whole fleet — orphan sweeps, lifecycle reconciliation — read a sandbox's
733
+ * ABSENCE from the list as evidence it is gone, so a truncated page and a small
734
+ * fleet must never be the same answer. Returning a short array with no signal
735
+ * makes them identical, and the caller that acts on that difference is the one
736
+ * deleting machines.
737
+ *
738
+ * A caller that sees `complete: false` has to leave every row alone, exactly as
739
+ * if it had never asked.
740
+ */
741
+ interface SandboxListPage {
742
+ sandboxes: SandboxInfo[];
743
+ /**
744
+ * False means THIS IS NOT THE WHOLE FLEET — for any reason, including a
745
+ * `limit` the caller set. A caller-imposed limit that stopped the walk while
746
+ * more sandboxes existed is still a truncated answer, and reporting it as
747
+ * complete is what made this flag useless at its only real consumer: a sweep
748
+ * that always passes a limit could never learn it had been truncated.
749
+ * Complete means the provider ran out, not that we stopped asking.
750
+ */
751
+ complete: boolean;
752
+ /** Provider requests made. Diagnostic — a fleet that suddenly costs 40 pages. */
753
+ pagesFetched: number;
754
+ /** Why it could not be finished, when it could not. */
755
+ error?: string;
756
+ }
757
+ /** Sandbox metadata and lifecycle info (capability: SandboxProvider.list, SandboxInstance.getInfo). */
758
+ interface SandboxInfo {
759
+ sandboxId: string;
760
+ /** The provider-neutral image/template the sandbox booted from. */
761
+ image: string;
762
+ name?: string;
763
+ metadata: Record<string, string>;
764
+ startedAt: string;
765
+ /** End time (undefined for running sandboxes). */
766
+ endAt?: string;
329
767
  }
330
768
  /** Command execution capabilities */
331
769
  interface SandboxCommands {
@@ -343,6 +781,27 @@ interface SandboxFiles {
343
781
  data: string | Buffer | ArrayBuffer | Uint8Array;
344
782
  }>): Promise<void>;
345
783
  makeDir(path: string): Promise<void>;
784
+ /**
785
+ * Upload a LOCAL file by path, without loading it into the process heap.
786
+ *
787
+ * write()/writeBatch() take the bytes as a value, so uploading a large
788
+ * artifact costs one full-size Buffer per concurrent upload — a caller doing
789
+ * many uploads at once pays that in RSS. This takes the path instead and lets
790
+ * the provider move the bytes its own cheapest way (a request body streamed
791
+ * off disk, or the vendor SDK's own path upload).
792
+ *
793
+ * OPTIONAL: a provider that has no cheaper path than "read it and send it"
794
+ * omits this, and uploadFileFromPath() falls back to write().
795
+ */
796
+ writeFromPath?(sandboxPath: string, localPath: string): Promise<void>;
797
+ /** Check whether a file or directory exists. */
798
+ exists?(path: string): Promise<boolean>;
799
+ /** List directory contents. */
800
+ list?(path: string): Promise<FileInfo[]>;
801
+ /** Delete a file or directory. */
802
+ remove?(path: string): Promise<void>;
803
+ /** Rename or move a file or directory. */
804
+ rename?(oldPath: string, newPath: string): Promise<void>;
346
805
  }
347
806
  /** Sandbox instance */
348
807
  interface SandboxInstance {
@@ -353,15 +812,62 @@ interface SandboxInstance {
353
812
  getHost(port: number): Promise<string>;
354
813
  kill(): Promise<void>;
355
814
  pause(): Promise<void>;
815
+ /** Whether the sandbox is currently running. */
816
+ isRunning?(): Promise<boolean>;
817
+ /** Sandbox metadata and timing. */
818
+ getInfo?(): Promise<SandboxInfo>;
356
819
  }
357
820
  /** Sandbox lifecycle management - providers implement this */
358
821
  interface SandboxProvider {
359
822
  /** Provider type identifier (e.g., "e2b") */
360
823
  readonly providerType: string;
361
- /** Human-readable provider name for logging */
824
+ /** Human-readable provider name for logging (e.g., "E2B") */
362
825
  readonly name?: string;
363
826
  create(options: SandboxCreateOptions): Promise<SandboxInstance>;
364
827
  connect(sandboxId: string, timeoutMs?: number): Promise<SandboxInstance>;
828
+ /**
829
+ * List sandboxes, paginating to exhaustion.
830
+ *
831
+ * `limit` bounds the number of items RETURNED, so a caller that wants one
832
+ * cheap page still asks for one; without it the answer is the whole fleet.
833
+ * It used to be first-page-only regardless, which silently truncated any
834
+ * account past a provider's page size.
835
+ *
836
+ * Errors throw. A caller that cannot treat a failed enumeration as an
837
+ * exception — because it reads absence as termination — wants `listAll`.
838
+ *
839
+ * OPTIONAL: all three first-party providers implement it, but the SDK never
840
+ * calls it, so requiring it would break third-party providers passed to
841
+ * .withSandbox() for no gain. Declared here so every provider that offers it
842
+ * offers the SAME signature.
843
+ */
844
+ list?(options?: SandboxListOptions): Promise<SandboxInfo[]>;
845
+ /**
846
+ * The fleet-bookkeeping counterpart to `list()`: paginates to exhaustion and
847
+ * NEVER throws.
848
+ *
849
+ * The difference is not error style, it is what a failure MEANS to the
850
+ * caller. Anything that reads a sandbox's absence as "terminated" cannot
851
+ * distinguish a provider that answered "nothing" from one that could not
852
+ * finish answering — and acting on that confusion mass-kills a live fleet. So
853
+ * a failure comes back as `complete: false` rather than as an exception the
854
+ * caller might catch and treat as an empty list.
855
+ *
856
+ * REQUIRED, unlike `list`. It was optional in the first cut, and that is
857
+ * precisely what let one provider keep silently truncating while this
858
+ * interface promised exhaustive listing — a provider that cannot answer "is
859
+ * this the whole fleet?" cannot be used for fleet bookkeeping at all, so the
860
+ * type refuses to let a fourth one ship without saying so.
861
+ *
862
+ * PARTIAL RESULTS ARE RETURNED, not discarded: `complete: false` with a
863
+ * non-empty `sandboxes` means "at least these, and there are more". Modal's
864
+ * own `listSandboxIds` takes the stricter line and returns an empty set on
865
+ * failure, on the grounds that partial results are worse than none for a
866
+ * terminal-state decision. Both are safe because `complete` is what callers
867
+ * branch on; the divergence is deliberate and noted here so nobody "fixes"
868
+ * one to match the other without deciding which rule they want.
869
+ */
870
+ listAll(options?: SandboxListOptions): Promise<SandboxListPage>;
365
871
  }
366
872
  /** Supported agent types (headless CLI agents only, no ACP) */
367
873
  type AgentType = "claude" | "codex" | "gemini" | "qwen" | "kimi" | "opencode" | "droid";
@@ -376,7 +882,7 @@ declare const AGENT_TYPES: {
376
882
  readonly DROID: "droid";
377
883
  };
378
884
  /** Workspace mode determines folder structure and system prompt */
379
- type WorkspaceMode = "knowledge" | "swe";
885
+ type WorkspaceMode = "knowledge" | "swe" | "task";
380
886
  /** Available skills that can be enabled */
381
887
  type SkillName = "pdf" | "dev-browser" | (string & {});
382
888
  /** Browser automation providers that can be enabled explicitly */
@@ -520,6 +1026,21 @@ interface SchemaValidationOptions {
520
1026
  * Validation mode preset definitions
521
1027
  */
522
1028
  declare const VALIDATION_PRESETS: Record<ValidationMode, Required<Omit<SchemaValidationOptions, "mode">>>;
1029
+ /**
1030
+ * Caller-minted gateway credential for external gateway mode.
1031
+ *
1032
+ * For callers that mint their own spend-capped key on an OpenAI-compatible
1033
+ * gateway. The credential is injected like direct mode and sealCredentials()
1034
+ * calls revoke() — sealing fails if revocation fails.
1035
+ */
1036
+ interface ExternalGatewayConfig {
1037
+ /** Spend-capped gateway API key minted by the caller */
1038
+ apiKey: string;
1039
+ /** OpenAI-compatible gateway base URL */
1040
+ baseUrl: string;
1041
+ /** Revoke the minted credential. Called by sealCredentials(); seal fails if this throws. */
1042
+ revoke: () => Promise<void>;
1043
+ }
523
1044
  /** Configuration passed to withAgent() */
524
1045
  interface AgentConfig {
525
1046
  /** Agent type (default: "claude") */
@@ -532,10 +1053,27 @@ interface AgentConfig {
532
1053
  oauthToken?: string;
533
1054
  /** Provider base URL for direct mode (default: provider env var or registry default) */
534
1055
  providerBaseUrl?: string;
1056
+ /**
1057
+ * Caller-minted revocable gateway credential. Mutually exclusive with
1058
+ * apiKey (gateway mode) and providerApiKey/providerBaseUrl (direct mode).
1059
+ */
1060
+ externalGateway?: ExternalGatewayConfig;
535
1061
  /** Model to use (optional, uses agent's default if omitted) */
536
1062
  model?: string;
537
1063
  /** Reasoning effort for models that support it */
538
1064
  reasoningEffort?: ReasoningEffort;
1065
+ /**
1066
+ * Context/completion ceiling for CLIs that must be told one (Kimi Code reads
1067
+ * it as `max_context_size` and sends it as the request's `max_tokens`).
1068
+ *
1069
+ * Set it to the model's real ceiling when driving a harness against a model
1070
+ * from another family — e.g. Kimi Code against `gpt-5.5` through an
1071
+ * OpenAI-compatible gateway, where an oversized `max_tokens` is rejected with
1072
+ * a 400. When set it is used verbatim. When omitted, the harness's own models
1073
+ * keep their registry value and any other model falls back to a conservative
1074
+ * 128000. Harnesses that never send a ceiling ignore it.
1075
+ */
1076
+ maxContextSize?: number;
539
1077
  }
540
1078
  /** Resolved agent config (output of resolution, not an extension of input) */
541
1079
  interface ResolvedAgentConfig {
@@ -546,15 +1084,29 @@ interface ResolvedAgentConfig {
546
1084
  isOAuth?: boolean;
547
1085
  /** File content for file-based OAuth (Codex) */
548
1086
  oauthFileContent?: string;
1087
+ /** External gateway mode: caller-minted revocable credential */
1088
+ externalGateway?: {
1089
+ revoke: () => Promise<void>;
1090
+ };
549
1091
  model?: string;
550
1092
  reasoningEffort?: ReasoningEffort;
1093
+ /** Caller-pinned context/completion ceiling; used verbatim when present */
1094
+ maxContextSize?: number;
551
1095
  }
552
1096
  /** Options for Agent constructor */
553
1097
  interface AgentOptions {
554
1098
  /** Sandbox provider (e.g., E2B) */
555
1099
  sandboxProvider?: SandboxProvider;
1100
+ /** Provider-neutral sandbox creation options forwarded on fresh creates. */
1101
+ sandboxCreateOptions?: SandboxCreateOptions;
556
1102
  /** Additional environment secrets */
557
1103
  secrets?: Record<string, string>;
1104
+ /** Dashboard-stored managed secrets exposed through opaque env vars. */
1105
+ managedSecrets?: {
1106
+ secrets: ManagedSecretRef[];
1107
+ apiKey: string;
1108
+ dashboardUrl?: string;
1109
+ };
558
1110
  /** Existing sandbox ID to connect to */
559
1111
  sandboxId?: string;
560
1112
  /** Working directory path */
@@ -589,6 +1141,11 @@ interface AgentOptions {
589
1141
  apiKey: string;
590
1142
  dashboardUrl?: string;
591
1143
  };
1144
+ /** Evolve-managed provider routing tokens for dashboard-stored BYOK provider keys. */
1145
+ providerRouting?: {
1146
+ apiKey: string;
1147
+ dashboardUrl?: string;
1148
+ };
592
1149
  /** Plugins/extensions to install in the sandbox user profile before first run */
593
1150
  plugins?: AgentPluginConfig[];
594
1151
  /** Skills to enable (e.g., ["pdf", "dev-browser"]) */
@@ -701,7 +1258,7 @@ interface RunCost {
701
1258
  runId: string;
702
1259
  /** 1-based chronological position in session */
703
1260
  index: number;
704
- /** Total cost in USD (includes platform margin) */
1261
+ /** Total cost in USD as billed to your Evolve account */
705
1262
  cost: number;
706
1263
  /** Token counts */
707
1264
  tokens: {
@@ -981,8 +1538,6 @@ interface IntegrationAccountDeleteResult {
981
1538
  *
982
1539
  * Single Agent class that uses registry lookup for agent-specific behavior.
983
1540
  * All agent differences are data (in registry), not code.
984
- *
985
- * Evidence: sdk-rewrite-v3.md Design Decisions section
986
1541
  */
987
1542
 
988
1543
  /**
@@ -997,9 +1552,10 @@ declare class Agent {
997
1552
  private sandbox?;
998
1553
  private hasRun;
999
1554
  private readonly workingDir;
1555
+ private readonly homeDir;
1000
1556
  private lastRunTimestamp?;
1001
1557
  private readonly registry;
1002
- /** Unified session ID — used for both observability (SessionLogger) and spend tracking (LiteLLM customer-id) */
1558
+ /** Unified session ID — used for both observability (SessionLogger) and spend tracking (gateway customer-id) */
1003
1559
  private sessionTag;
1004
1560
  /** Previous session tag — preserved across kill()/setSession() so cost queries still work */
1005
1561
  private previousSessionTag?;
@@ -1014,6 +1570,10 @@ declare class Agent {
1014
1570
  private agentState;
1015
1571
  private droidSessionId?;
1016
1572
  private managedBrowserSession?;
1573
+ private providerRuntimeToken?;
1574
+ private managedSecretRuntimeToken?;
1575
+ private managedSecretProxy?;
1576
+ private credentialsSealed;
1017
1577
  private readonly skills?;
1018
1578
  private readonly storage?;
1019
1579
  private lastCheckpointId?;
@@ -1037,13 +1597,52 @@ declare class Agent {
1037
1597
  * Get or create sandbox instance
1038
1598
  */
1039
1599
  getSandbox(callbacks?: StreamCallbacks): Promise<SandboxInstance>;
1600
+ /**
1601
+ * The `max_context_size` Kimi Code is told, which it sends as the request's
1602
+ * `max_tokens`. Three cases, in order:
1603
+ *
1604
+ * 1. `maxContextSize` on the agent config wins verbatim — the caller knows
1605
+ * the model's real ceiling.
1606
+ * 2. A model the kimi registry entry itself declares keeps that model's
1607
+ * own ceiling when the row declares one (K3: 1048576), else the
1608
+ * harness default (262144).
1609
+ * 3. Any other model — e.g. driving Kimi Code against `gpt-5.5` through an
1610
+ * OpenAI-compatible gateway — gets the conservative constant, because
1611
+ * 262144 is above that model's ceiling and the gateway answers 400.
1612
+ *
1613
+ * Both kimi wiring paths (KIMI_MODEL_MAX_CONTEXT_SIZE and the config.toml
1614
+ * `max_context_size`) resolve it here, so they can never disagree.
1615
+ */
1616
+ private resolveKimiMaxContextSize;
1617
+ /**
1618
+ * The effort this agent's runs stamp on the wire: the caller's
1619
+ * reasoningEffort when given, else the harness's registry-pinned default.
1620
+ * Every command/env/config build path reads this, so an omitted effort is
1621
+ * an explicit stamp of the pin — never the vendor's silent default.
1622
+ */
1623
+ private reasoningEffort;
1624
+ /**
1625
+ * Kimi Code model envs for direct-style credential injection (direct mode
1626
+ * and externalGateway mode). Kimi Code reads KIMI_MODEL_* — the registry's
1627
+ * KIMI_API_KEY/KIMI_BASE_URL are SDK-facing inputs the CLI never reads.
1628
+ */
1629
+ private buildKimiDirectModelEnvs;
1040
1630
  /**
1041
1631
  * Build environment variables for sandbox
1042
1632
  */
1043
1633
  private buildEnvironmentVariables;
1634
+ private buildSandboxCreateOptions;
1635
+ private validatedUserSecretsForEnvironment;
1044
1636
  private ensureManagedBrowserSession;
1637
+ private ensureProviderRuntimeToken;
1638
+ private bindProviderRuntimeToken;
1045
1639
  private setupManagedBrowser;
1046
1640
  private closeManagedBrowserSession;
1641
+ private closeProviderRuntimeToken;
1642
+ private ensureManagedSecretRuntimeToken;
1643
+ private bindManagedSecretRuntimeToken;
1644
+ private setupManagedSecretEgress;
1645
+ private closeManagedSecretRuntimeToken;
1047
1646
  /**
1048
1647
  * Build the inline gateway config JSON for agents using gatewayConfigEnv
1049
1648
  * (e.g., OpenCode OPENCODE_CONFIG_CONTENT). Centralizes the provider config
@@ -1056,6 +1655,14 @@ declare class Agent {
1056
1655
  * Source-verified: model.headers → provider.ts:1061 → llm.ts:221 → HTTP request.
1057
1656
  */
1058
1657
  private buildGatewayConfigJson;
1658
+ private activeProviderRuntimeToken;
1659
+ private requireActiveProviderRuntimeToken;
1660
+ private ensureSessionLogger;
1661
+ private flushSessionLoggerWithTimeout;
1662
+ private requiresPreRunDashboardIngest;
1663
+ private providerRuntimeHeaderUpdates;
1664
+ private shouldExposeProviderRuntimeTokenEnv;
1665
+ private buildProviderRuntimeProcessEnvs;
1059
1666
  /**
1060
1667
  * Build per-run env overrides for spend tracking.
1061
1668
  * Merges session + run headers into the custom headers env var,
@@ -1063,9 +1670,11 @@ declare class Agent {
1063
1670
  * Passed to spawn() so each .run() gets a unique run tag.
1064
1671
  */
1065
1672
  private buildRunEnvs;
1673
+ private writeCodexGatewayProviderConfig;
1066
1674
  private captureDroidSession;
1067
1675
  private extractDroidSessionId;
1068
1676
  private findDroidSessionId;
1677
+ private droidSessionStatePath;
1069
1678
  private loadDroidSessionState;
1070
1679
  private writeDroidSessionState;
1071
1680
  private resolveGatewayModel;
@@ -1074,7 +1683,9 @@ declare class Agent {
1074
1683
  * Agent-specific authentication setup
1075
1684
  */
1076
1685
  private setupAgentAuth;
1686
+ private writeGeminiGatewayAuthSettings;
1077
1687
  private setupAgentPlugins;
1688
+ private assertProviderRuntimeDoesNotExposeGatewayKey;
1078
1689
  /**
1079
1690
  * Setup workspace structure and files
1080
1691
  *
@@ -1106,11 +1717,25 @@ declare class Agent {
1106
1717
  *
1107
1718
  * Streams output via callbacks, returns final response.
1108
1719
  */
1720
+ /** Per-run TOML provider spend tracking (Kimi): write provider with
1721
+ * custom_headers before spawning. CLI reads config at startup, so each run
1722
+ * gets a fresh write. Extracted so the effort pin on this path is testable. */
1723
+ private writeKimiPerRunConfig;
1109
1724
  run(options: RunOptions, callbacks?: StreamCallbacks): Promise<AgentResponse>;
1110
1725
  /**
1111
1726
  * Execute arbitrary command in sandbox
1112
1727
  */
1113
1728
  executeCommand(command: string, options?: ExecuteCommandOptions, callbacks?: StreamCallbacks): Promise<AgentResponse>;
1729
+ /**
1730
+ * Permanently revoke the model capability attached to this sandbox.
1731
+ * This is intentionally fail-closed: configurations that may have placed
1732
+ * other credentials in the sandbox cannot claim to be sealed.
1733
+ */
1734
+ sealCredentials(): Promise<void>;
1735
+ /** Whether sealCredentials() has completed — the sandbox holds no revocable model credential. */
1736
+ isSealed(): boolean;
1737
+ /** Collect caller-declared files or directories after the credential boundary. */
1738
+ collectArtifacts(paths: string[]): Promise<FileMap>;
1114
1739
  /**
1115
1740
  * Upload context files (to context/ folder)
1116
1741
  */
@@ -1119,6 +1744,24 @@ declare class Agent {
1119
1744
  * Upload files to working directory
1120
1745
  */
1121
1746
  uploadFiles(files: FileMap): Promise<void>;
1747
+ /**
1748
+ * Upload one LOCAL file into the sandbox by path, without holding its bytes
1749
+ * in the process heap.
1750
+ *
1751
+ * uploadFiles() takes a FileMap — the bytes as a value — so a caller
1752
+ * uploading a large artifact pays one full-size Buffer per concurrent
1753
+ * upload. This takes the local path instead: the provider streams it off
1754
+ * disk (or hands its own SDK the path), so peak memory is a chunk rather
1755
+ * than the file. Use it for anything big enough that N concurrent uploads
1756
+ * would matter; uploadFiles() stays the right call for small content.
1757
+ *
1758
+ * `sandboxPath` follows the uploadFiles() convention: absolute paths are
1759
+ * used as-is, relative paths resolve under the working directory.
1760
+ *
1761
+ * Providers that expose no cheaper path than a plain write fall back to
1762
+ * reading the file — correct everywhere, cheap where the provider helps.
1763
+ */
1764
+ uploadFileFromPath(sandboxPath: string, localPath: string): Promise<void>;
1122
1765
  /**
1123
1766
  * Get output files from output/ folder with optional schema validation
1124
1767
  *
@@ -1146,250 +1789,77 @@ declare class Agent {
1146
1789
  * Set session (sandbox ID) to connect to
1147
1790
  *
1148
1791
  * When reconnecting to an existing sandbox, we assume the agent
1149
- * may have already run commands, so we set hasRun=true to use
1150
- * the continue/resume command template instead of first-run.
1151
- */
1152
- setSession(sandboxId: string): Promise<void>;
1153
- /**
1154
- * Pause sandbox
1155
- */
1156
- pause(callbacks?: StreamCallbacks): Promise<void>;
1157
- /**
1158
- * Resume sandbox
1159
- */
1160
- resume(callbacks?: StreamCallbacks): Promise<void>;
1161
- /**
1162
- * Interrupt active command without killing the sandbox.
1163
- */
1164
- interrupt(callbacks?: StreamCallbacks): Promise<boolean>;
1165
- /**
1166
- * Get current runtime status for sandbox and agent.
1167
- */
1168
- status(): SessionStatus;
1169
- /**
1170
- * Kill sandbox (terminates all processes)
1171
- */
1172
- kill(callbacks?: StreamCallbacks): Promise<void>;
1173
- /**
1174
- * Get host URL for a port
1175
- */
1176
- getHost(port: number): Promise<string>;
1177
- /**
1178
- * Get agent type
1179
- */
1180
- getAgentType(): AgentType;
1181
- /**
1182
- * Get current session tag.
1183
- * Returns null if no active session (before sandbox creation or after kill()).
1184
- * Used for both observability (dashboard traces) and spend tracking (LiteLLM customer-id).
1185
- */
1186
- getSessionTag(): string | null;
1187
- /**
1188
- * Get current session timestamp
1189
- *
1190
- * Returns null if no session has started (run() not called yet).
1191
- */
1192
- getSessionTimestamp(): string | null;
1193
- /**
1194
- * Flush pending observability events without closing the session.
1195
- */
1196
- flushObservability(): Promise<void>;
1197
- /**
1198
- * Tear down the session logger and rotate the session tag.
1199
- * Preserves `previousSessionTag` only if this session had actual activity,
1200
- * so a double kill() or no-op lifecycle call doesn't clobber the real tag.
1201
- * @internal
1202
- */
1203
- private rotateSession;
1204
- /**
1205
- * Fetch spend data from dashboard API.
1206
- * @internal
1207
- */
1208
- private fetchSpend;
1209
- /**
1210
- * Resolve the session tag for cost queries.
1211
- * Uses the active session tag, or falls back to the previous tag after kill()/setSession().
1212
- * @internal
1213
- */
1214
- private resolveSpendTag;
1215
- /**
1216
- * Normalize run payloads for compatibility with older dashboard responses.
1217
- * Older responses may omit `asOf`, `isComplete`, or `truncated` inside `runs[]`.
1218
- * @internal
1219
- */
1220
- private normalizeRunCost;
1221
- /**
1222
- * Normalize session payloads so all `runs[]` conform to `RunCost`.
1223
- * @internal
1224
- */
1225
- private normalizeSessionCost;
1226
- /**
1227
- * Get cost breakdown for the current session (all runs).
1228
- *
1229
- * Queries the dashboard API which proxies to LiteLLM spend logs.
1230
- * Cost data has ~60s latency due to gateway batch writes.
1231
- * Also works after kill() for the most recent session only.
1232
- *
1233
- * Requires gateway mode (EVOLVE_API_KEY).
1234
- */
1235
- getSessionCost(): Promise<SessionCost>;
1236
- /**
1237
- * Get cost for a specific run by ID or index.
1238
- *
1239
- * @param run - Either `{ runId: string }` or `{ index: number }` (1-based, negative = from end)
1240
- *
1241
- * Also works after kill() for the most recent session only.
1242
- * Requires gateway mode (EVOLVE_API_KEY).
1243
- */
1244
- getRunCost(run: {
1245
- runId: string;
1246
- } | {
1247
- index: number;
1248
- }): Promise<RunCost>;
1249
- }
1250
-
1251
- /**
1252
- * Claude JSONL → ACP-style events parser.
1253
- *
1254
- * Native schema source (@anthropic-ai/claude-agent-sdk):
1255
- * MANUS-API/KNOWLEDGE/claude-agent-sdk/cc_sdk_typescript.md
1256
- * (SDKMessage, SDKAssistantMessage, SDKPartialAssistantMessage, Tool Input/Output types)
1257
- *
1258
- * Conversion logic reference:
1259
- * MANUS-API/KNOWLEDGE/claude-code-acp/src/tools.ts
1260
- * (toolInfoFromToolUse, toolUpdateFromToolResult)
1261
- *
1262
- * ACP output schema:
1263
- * MANUS-API/KNOWLEDGE/acp-typescript-sdk/src/schema/types.gen.ts
1264
- */
1265
-
1266
- /**
1267
- * Create a Claude parser instance with its own isolated cache.
1268
- * Each Evolve instance should create its own parser for proper isolation.
1269
- */
1270
- declare function createClaudeParser(): (jsonLine: string) => OutputEvent[] | null;
1271
-
1272
- /**
1273
- * Codex JSONL → ACP-style events parser.
1274
- *
1275
- * Native schema: codex-rs/exec/src/exec_events.rs
1276
- * - ThreadEvent: thread.started, turn.started, turn.completed, item.started, item.updated, item.completed
1277
- * - ThreadItemDetails: AgentMessage, Reasoning, CommandExecution, FileChange, McpToolCall, WebSearch, TodoList, Error
1278
- *
1279
- * ACP output: acp-typescript-sdk/src/schema/types.gen.ts
1280
- * - SessionUpdate: agent_message_chunk, agent_thought_chunk, tool_call, tool_call_update, plan
1281
- *
1282
- * Event mapping:
1283
- * reasoning → agent_thought_chunk (exec_events.rs:134 ReasoningItem { text })
1284
- * agent_message → agent_message_chunk (exec_events.rs:129 AgentMessageItem { text })
1285
- * mcp_tool_call → tool_call/update (exec_events.rs:215 McpToolCallItem)
1286
- * command_execution → tool_call/update (exec_events.rs:151 CommandExecutionItem)
1287
- * file_change → tool_call (exec_events.rs:176 FileChangeItem)
1288
- * todo_list → plan (exec_events.rs:245 TodoListItem { items: TodoItem[] })
1289
- * web_search → tool_call (exec_events.rs:227 WebSearchItem { query })
1290
- */
1291
-
1292
- /**
1293
- * Create a Codex parser instance.
1294
- */
1295
- declare function createCodexParser(): (jsonLine: string) => OutputEvent[] | null;
1296
-
1297
- /**
1298
- * Droid exec parser.
1299
- *
1300
- * Supports the documented headless `--output-format stream-json` lines and the
1301
- * raw `stream-jsonrpc` notification envelope used by Droid's low-level SDK.
1302
- */
1303
-
1304
- declare function createDroidParser(): (jsonLine: string) => OutputEvent[] | null;
1305
-
1306
- /**
1307
- * Gemini JSONL → ACP-style events parser.
1308
- *
1309
- * Native schema (gemini --output-format stream-json):
1310
- * gemini-cli/packages/core/src/output/types.ts
1311
- *
1312
- * Gemini events (types.ts:29-36 JsonStreamEventType):
1313
- * - "init" → types.ts:43-47 InitEvent { session_id, model }
1314
- * - "message" → types.ts:49-54 MessageEvent { role, content, delta? }
1315
- * - "tool_use" → types.ts:56-61 ToolUseEvent { tool_name, tool_id, parameters }
1316
- * - "tool_result" → types.ts:63-72 ToolResultEvent { tool_id, status, output?, error? }
1317
- * - "error" → types.ts:74-78 ErrorEvent { severity, message }
1318
- * - "result" → types.ts:91-99 ResultEvent { status, error?, stats? }
1319
- *
1320
- * ACP output: acp-typescript-sdk/src/schema/types.gen.ts:2449-2464
1321
- */
1322
-
1323
- /**
1324
- * Create a Gemini parser instance.
1325
- */
1326
- declare function createGeminiParser(): (jsonLine: string) => OutputEvent[] | null;
1327
-
1328
- /**
1329
- * Qwen NDJSON → ACP-style events parser.
1330
- *
1331
- * Native schema: KNOWLEDGE/qwen-code/packages/sdk-typescript/src/types/protocol.ts
1332
- * ACP schema: KNOWLEDGE/acp-typescript-sdk/src/schema/types.gen.ts
1333
- *
1334
- * Qwen NDJSON message types (protocol.ts:428-433):
1335
- * - type: "assistant" → SDKAssistantMessage (protocol.ts:102-108)
1336
- * - type: "stream_event" → SDKPartialAssistantMessage (protocol.ts:225-231)
1337
- * - type: "user" → SDKUserMessage (protocol.ts:93-100)
1338
- * - type: "system" → SDKSystemMessage (skipped)
1339
- * - type: "result" → SDKResultMessage (skipped)
1340
- *
1341
- * ContentBlock types (protocol.ts:72-76):
1342
- * - TextBlock (protocol.ts:43-48): { type: 'text', text: string }
1343
- * - ThinkingBlock (protocol.ts:49-54): { type: 'thinking', thinking: string }
1344
- * - ToolUseBlock (protocol.ts:56-62): { type: 'tool_use', id, name, input }
1345
- * - ToolResultBlock (protocol.ts:64-70): { type: 'tool_result', tool_use_id, content?, is_error? }
1346
- *
1347
- * StreamEvent types (protocol.ts:218-223):
1348
- * - message_start, content_block_start, content_block_delta, content_block_stop, message_stop
1349
- */
1350
-
1351
- /**
1352
- * Stateless parser function (creates new parser per call).
1353
- * Use createQwenParser() for stateful streaming parsing.
1354
- *
1355
- * @param line - Single line of NDJSON from qwen CLI
1356
- * @returns Array of OutputEvent objects, or null if line couldn't be parsed
1357
- */
1358
- declare function parseQwenOutput(line: string): OutputEvent[] | null;
1359
-
1360
- /**
1361
- * Unified Parser Entry Point
1362
- *
1363
- * Routes NDJSON lines to the appropriate agent-specific parser.
1364
- * Simple line-based parsing - no buffering needed since CLIs output complete JSON per line.
1365
- */
1366
-
1367
- /** Parser function type */
1368
- type AgentParser = (jsonLine: string) => OutputEvent[] | null;
1369
- /**
1370
- * Create a parser instance for the given agent type.
1371
- * Each Evolve instance should create its own parser for proper isolation.
1372
- *
1373
- * @param agentType - The agent type to create a parser for
1374
- * @returns Parser function that takes NDJSON lines and returns OutputEvents
1375
- */
1376
- declare function createAgentParser(agentType: AgentType): AgentParser;
1377
- /**
1378
- * Parse a single NDJSON line from any agent (creates new parser per call - use createAgentParser for efficiency)
1379
- *
1380
- * @param agentType - The agent type to parse for
1381
- * @param line - Single line of NDJSON output
1382
- * @returns Array of OutputEvent objects, or null if line couldn't be parsed
1383
- */
1384
- declare function parseNdjsonLine(agentType: AgentType, line: string): OutputEvent[] | null;
1385
- /**
1386
- * Parse multiple NDJSON lines (convenience wrapper)
1387
- *
1388
- * @param agentType - The agent type to parse for
1389
- * @param output - Multi-line NDJSON output
1390
- * @returns Array of all parsed OutputEvent objects
1391
- */
1392
- declare function parseNdjsonOutput(agentType: AgentType, output: string): OutputEvent[];
1792
+ * may have already run commands, so we set hasRun=true to use
1793
+ * the continue/resume command template instead of first-run.
1794
+ */
1795
+ setSession(sandboxId: string): Promise<void>;
1796
+ /**
1797
+ * Pause sandbox
1798
+ */
1799
+ pause(callbacks?: StreamCallbacks): Promise<void>;
1800
+ /**
1801
+ * Resume sandbox
1802
+ */
1803
+ resume(callbacks?: StreamCallbacks): Promise<void>;
1804
+ /**
1805
+ * Interrupt active command without killing the sandbox.
1806
+ */
1807
+ interrupt(callbacks?: StreamCallbacks): Promise<boolean>;
1808
+ /**
1809
+ * Get current runtime status for sandbox and agent.
1810
+ */
1811
+ status(): SessionStatus;
1812
+ /**
1813
+ * Kill sandbox (terminates all processes)
1814
+ */
1815
+ kill(callbacks?: StreamCallbacks): Promise<void>;
1816
+ /**
1817
+ * Get host URL for a port
1818
+ */
1819
+ getHost(port: number): Promise<string>;
1820
+ /**
1821
+ * Get agent type
1822
+ */
1823
+ getAgentType(): AgentType;
1824
+ /**
1825
+ * Get current session tag.
1826
+ * Returns null if no active session (before sandbox creation or after kill()).
1827
+ * Used for both observability (dashboard traces) and spend tracking (gateway customer-id).
1828
+ */
1829
+ getSessionTag(): string | null;
1830
+ /**
1831
+ * Get current session timestamp
1832
+ *
1833
+ * Returns null if no session has started (run() not called yet).
1834
+ */
1835
+ getSessionTimestamp(): string | null;
1836
+ /**
1837
+ * Flush pending observability events without closing the session.
1838
+ */
1839
+ flushObservability(): Promise<void>;
1840
+ /**
1841
+ * Get cost breakdown for the current session (all runs).
1842
+ *
1843
+ * Cost data can lag live usage by about a minute.
1844
+ * Also works after kill() for the most recent session only.
1845
+ *
1846
+ * Requires gateway mode (EVOLVE_API_KEY).
1847
+ */
1848
+ getSessionCost(): Promise<SessionCost>;
1849
+ /**
1850
+ * Get cost for a specific run by ID or index.
1851
+ *
1852
+ * @param run - Either `{ runId: string }` or `{ index: number }` (1-based, negative = from end)
1853
+ *
1854
+ * Also works after kill() for the most recent session only.
1855
+ * Requires gateway mode (EVOLVE_API_KEY).
1856
+ */
1857
+ getRunCost(run: {
1858
+ runId: string;
1859
+ } | {
1860
+ index: number;
1861
+ }): Promise<RunCost>;
1862
+ }
1393
1863
 
1394
1864
  declare const BROWSER_LOGIN_MCP_SERVER_NAME = "browser-login";
1395
1865
  interface BrowserCredentialMetadata {
@@ -1472,52 +1942,6 @@ declare class BrowserProfilesClient {
1472
1942
  }
1473
1943
  declare function browserProfiles(config?: BrowserProfilesClientConfig): BrowserProfilesClient;
1474
1944
 
1475
- /**
1476
- * Evolve events
1477
- *
1478
- * Runtime streams:
1479
- * - stdout: Raw NDJSON lines
1480
- * - stderr: Process stderr
1481
- * - content: Parsed OutputEvent
1482
- * - lifecycle: Sandbox/agent lifecycle transitions
1483
- */
1484
- interface EvolveEvents {
1485
- stdout: (chunk: string) => void;
1486
- stderr: (chunk: string) => void;
1487
- content: (event: OutputEvent) => void;
1488
- lifecycle: (event: LifecycleEvent) => void;
1489
- }
1490
- interface EvolveConfig {
1491
- agent?: AgentConfig;
1492
- sandbox?: SandboxProvider;
1493
- workingDirectory?: string;
1494
- workspaceMode?: WorkspaceMode;
1495
- secrets?: Record<string, string>;
1496
- sandboxId?: string;
1497
- systemPrompt?: string;
1498
- context?: FileMap;
1499
- files?: FileMap;
1500
- mcpServers?: Record<string, McpServerConfig>;
1501
- /** Browser automation provider to enable explicitly */
1502
- browser?: BrowserConfig;
1503
- /** Browser login MCP setup for managed remote agent-browser runs */
1504
- browserCredentials?: BrowserCredentialsConfig;
1505
- /** Agent plugins/extensions to install before first run */
1506
- plugins?: AgentPluginConfig[];
1507
- /** Skills to enable (e.g., ["pdf", "dev-browser"]) */
1508
- skills?: SkillName[];
1509
- /** Schema for structured output (Zod or JSON Schema, auto-detected) */
1510
- schema?: z.ZodType<unknown> | JsonSchema;
1511
- /** Validation options for JSON Schema (ignored for Zod) */
1512
- schemaOptions?: SchemaValidationOptions;
1513
- sessionTagPrefix?: string;
1514
- /** Observability metadata for trace grouping (generic key-value, domain-agnostic) */
1515
- observability?: Record<string, unknown>;
1516
- /** Managed integrations config */
1517
- integrations?: IntegrationsSetup;
1518
- /** Storage configuration for checkpointing */
1519
- storage?: StorageConfig;
1520
- }
1521
1945
  /**
1522
1946
  * Evolve orchestrator with builder pattern
1523
1947
  *
@@ -1551,6 +1975,11 @@ declare class Evolve extends EventEmitter {
1551
1975
  * Configure sandbox provider
1552
1976
  */
1553
1977
  withSandbox(provider?: SandboxProvider): this;
1978
+ /**
1979
+ * Configure provider-neutral options used whenever Evolve creates a sandbox.
1980
+ * Evolve-owned runtime variables override conflicting env entries.
1981
+ */
1982
+ withSandboxCreateOptions(options: SandboxCreateOptions): this;
1554
1983
  /**
1555
1984
  * Set working directory path
1556
1985
  */
@@ -1559,12 +1988,19 @@ declare class Evolve extends EventEmitter {
1559
1988
  * Set workspace mode
1560
1989
  * - "knowledge": Creates context/, scripts/, temp/, output/ folders
1561
1990
  * - "swe": Same as knowledge + repo/ folder for code repositories
1991
+ * - "task": Leaves the task-owned working directory untouched
1562
1992
  */
1563
1993
  withWorkspaceMode(mode: WorkspaceMode): this;
1564
1994
  /**
1565
1995
  * Add environment secrets
1566
1996
  */
1567
1997
  withSecrets(secrets: Record<string, string>): this;
1998
+ /**
1999
+ * Attach Dashboard-stored managed secrets to the sandbox.
2000
+ *
2001
+ * The sandbox receives opaque env var values; raw secret values stay server-side.
2002
+ */
2003
+ withManagedSecrets(secrets: ManagedSecretRef[]): this;
1568
2004
  /**
1569
2005
  * Connect to existing session
1570
2006
  */
@@ -1651,11 +2087,6 @@ declare class Evolve extends EventEmitter {
1651
2087
  * Set session tag prefix for observability
1652
2088
  */
1653
2089
  withSessionTagPrefix(prefix: string): this;
1654
- /**
1655
- * @internal Set observability metadata for trace grouping.
1656
- * Used internally by Swarm - not part of public API.
1657
- */
1658
- withObservability(meta: Record<string, unknown>): this;
1659
2090
  /**
1660
2091
  * Enable Evolve-managed integrations.
1661
2092
  *
@@ -1693,6 +2124,8 @@ declare class Evolve extends EventEmitter {
1693
2124
  static browserCredentials: typeof browserCredentials;
1694
2125
  /** Static browser profile client for listing and deleting reusable browser profiles. */
1695
2126
  static browserProfiles: typeof browserProfiles;
2127
+ /** Static managed secrets client for listing Dashboard-stored secret metadata. */
2128
+ static managedSecrets: typeof managedSecrets;
1696
2129
  /**
1697
2130
  * Initialize agent on first use
1698
2131
  */
@@ -1721,6 +2154,21 @@ declare class Evolve extends EventEmitter {
1721
2154
  timeoutMs?: number;
1722
2155
  background?: boolean;
1723
2156
  }): Promise<AgentResponse>;
2157
+ /**
2158
+ * Create and fully initialize the configured sandbox without starting an
2159
+ * agent command. Durable orchestrators use this to persist the sandbox ID
2160
+ * before handing execution to the agent.
2161
+ */
2162
+ prepareSandbox(): Promise<string>;
2163
+ /**
2164
+ * Irreversibly revoke Evolve-managed runtime credentials for this sandbox.
2165
+ * Agent runs are disabled afterward; credential-free commands remain available.
2166
+ */
2167
+ sealCredentials(): Promise<void>;
2168
+ /** Whether sealCredentials() has completed for the active sandbox. */
2169
+ isSealed(): boolean;
2170
+ /** Collect files or directories from the working directory after credentials are sealed. */
2171
+ collectArtifacts(paths: string[]): Promise<FileMap>;
1724
2172
  /**
1725
2173
  * Interrupt active process without killing sandbox.
1726
2174
  */
@@ -1733,6 +2181,15 @@ declare class Evolve extends EventEmitter {
1733
2181
  * Upload files to workspace (runtime - immediate upload)
1734
2182
  */
1735
2183
  uploadFiles(files: FileMap): Promise<void>;
2184
+ /**
2185
+ * Upload one LOCAL file into the sandbox by path, streaming rather than
2186
+ * buffering it (runtime - immediate upload).
2187
+ *
2188
+ * The memory-bounded counterpart to uploadFiles(): use it when the file is
2189
+ * large enough that holding it in the heap — once per concurrent upload —
2190
+ * would matter.
2191
+ */
2192
+ uploadFileFromPath(sandboxPath: string, localPath: string): Promise<void>;
1736
2193
  /**
1737
2194
  * Get output files from output/ folder with optional schema validation
1738
2195
  *
@@ -1924,7 +2381,7 @@ interface SwarmConfig {
1924
2381
  /** Per-worker timeout in ms (default: 1 hour) */
1925
2382
  timeoutMs?: number;
1926
2383
  /** Workspace mode (default: SDK default 'knowledge') */
1927
- workspaceMode?: WorkspaceMode;
2384
+ workspaceMode?: Exclude<WorkspaceMode, "task">;
1928
2385
  /** Default retry configuration for all operations (per-operation config takes precedence) */
1929
2386
  retry?: RetryConfig;
1930
2387
  /** Default MCP servers for all operations (per-operation config takes precedence) */
@@ -2125,7 +2582,7 @@ interface VerifyInfo {
2125
2582
  type ItemInput = FileMap | SwarmResult<unknown>;
2126
2583
  type PromptFn = (files: FileMap, index: number) => string;
2127
2584
  type Prompt = string | PromptFn;
2128
- /** @internal Pipeline context for observability (set by Pipeline, not user) */
2585
+ /** Pipeline context for observability (set by Pipeline, not user) — exported from the package root */
2129
2586
  interface PipelineContext {
2130
2587
  pipelineRunId: string;
2131
2588
  pipelineStepIndex: number;
@@ -2140,8 +2597,6 @@ interface MapParams<T> {
2140
2597
  name?: string;
2141
2598
  /** Optional system prompt */
2142
2599
  systemPrompt?: string;
2143
- /** @internal Pipeline context (set by Pipeline, not user) */
2144
- _pipelineContext?: PipelineContext;
2145
2600
  /** Schema for structured output (Zod or JSON Schema) */
2146
2601
  schema?: z.ZodType<T> | JsonSchema;
2147
2602
  /** Validation options for JSON Schema (ignored for Zod) */
@@ -2171,8 +2626,6 @@ interface FilterParams<T> {
2171
2626
  prompt: string;
2172
2627
  /** Optional operation name for observability */
2173
2628
  name?: string;
2174
- /** @internal Pipeline context (set by Pipeline, not user) */
2175
- _pipelineContext?: PipelineContext;
2176
2629
  /** Schema for structured output (Zod or JSON Schema) */
2177
2630
  schema: z.ZodType<T> | JsonSchema;
2178
2631
  /** Validation options for JSON Schema (ignored for Zod) */
@@ -2206,8 +2659,6 @@ interface ReduceParams<T> {
2206
2659
  name?: string;
2207
2660
  /** Optional system prompt */
2208
2661
  systemPrompt?: string;
2209
- /** @internal Pipeline context (set by Pipeline, not user) */
2210
- _pipelineContext?: PipelineContext;
2211
2662
  /** Schema for structured output (Zod or JSON Schema) */
2212
2663
  schema?: z.ZodType<T> | JsonSchema;
2213
2664
  /** Validation options for JSON Schema (ignored for Zod) */
@@ -2475,7 +2926,7 @@ interface ReduceConfig<T> extends BaseStepConfig {
2475
2926
  /** Retry configuration */
2476
2927
  retry?: RetryConfig<ReduceResult<T>>;
2477
2928
  }
2478
- /** @internal Step representation */
2929
+ /** Step representation — reaches the public surface through Pipeline's constructor */
2479
2930
  type Step = {
2480
2931
  type: "map";
2481
2932
  config: MapConfig<unknown>;
@@ -2486,7 +2937,7 @@ type Step = {
2486
2937
  type: "reduce";
2487
2938
  config: ReduceConfig<unknown>;
2488
2939
  };
2489
- /** @internal Step type literal */
2940
+ /** Step type literal — reaches the public surface as StepResult.type */
2490
2941
  type StepType = "map" | "filter" | "reduce";
2491
2942
  /** Result of a single pipeline step */
2492
2943
  interface StepResult<T = unknown> {
@@ -2668,176 +3119,203 @@ declare class TerminalPipeline<T> extends Pipeline<T> {
2668
3119
  }
2669
3120
 
2670
3121
  /**
2671
- * Agent Registry
2672
- *
2673
- * Single source of truth for agent-specific behavior.
2674
- * All differences between agents are data, not code.
2675
- *
2676
- * Evidence: sdk-rewrite-v3.md Agent Registry section
2677
- */
2678
-
2679
- /** Model configuration */
2680
- interface ModelInfo {
2681
- /** Model alias (short name used with --model) */
2682
- alias: string;
2683
- /** Full model ID */
2684
- modelId: string;
2685
- /** What this model is best for */
2686
- description: string;
2687
- }
2688
- /** MCP configuration for an agent */
2689
- interface McpConfigInfo {
2690
- /** Settings directory (e.g., "~/.claude") */
2691
- settingsDir: string;
2692
- /** Config filename (e.g., "settings.json" or "config.toml") */
2693
- filename: string;
2694
- /** Config format */
2695
- format: "json" | "toml";
2696
- /** Whether to use workingDir for project-level config (Claude only) */
2697
- projectConfig?: boolean;
2698
- }
2699
- /** Options for building agent commands */
2700
- interface BuildCommandOptions {
2701
- prompt: string;
2702
- model: string;
2703
- isResume: boolean;
2704
- sessionId?: string;
2705
- reasoningEffort?: string;
2706
- isDirectMode?: boolean;
2707
- /** Skills enabled for this run */
2708
- skills?: string[];
2709
- }
2710
- interface AgentRegistryEntry {
2711
- /** Sandbox image/template identifier (provider maps to its own concept) */
2712
- image: string;
2713
- /** Environment variable name for API key */
2714
- apiKeyEnv: string;
2715
- /** Environment variable name for OAuth (file path or token depending on agent) */
2716
- oauthEnv?: string;
2717
- /** OAuth credentials filename (e.g., "auth.json" for Codex, "oauth_creds.json" for Gemini) */
2718
- oauthFileName?: string;
2719
- /** Environment variable to set when OAuth is active (e.g., GOOGLE_GENAI_USE_GCA=true for Gemini) */
2720
- oauthActivationEnv?: {
2721
- key: string;
2722
- value: string;
2723
- };
2724
- /** Environment variable name for base URL, if this CLI supports one */
2725
- baseUrlEnv?: string;
2726
- /** Default model alias */
2727
- defaultModel: string;
2728
- /** Available models for this agent */
2729
- models: ModelInfo[];
2730
- /** System prompt filename (e.g., "CLAUDE.md") */
2731
- systemPromptFile: string;
2732
- /** MCP configuration */
2733
- mcpConfig: McpConfigInfo;
2734
- /** Build the CLI command for this agent */
2735
- buildCommand: (opts: BuildCommandOptions) => string;
2736
- /** Extra setup step (e.g., codex login) */
2737
- setupCommand?: string;
2738
- /** Gateway path prefix for CLIs that use a provider-native passthrough endpoint */
2739
- gatewayPath?: string;
2740
- /** Default base URL for direct mode (only needed if provider requires specific endpoint, e.g., Qwen → Dashscope) */
2741
- defaultBaseUrl?: string;
2742
- /** Available beta headers for this agent (for reference) */
2743
- availableBetas?: Record<string, string>;
2744
- /** Skills configuration for this agent */
2745
- skillsConfig: SkillsConfig;
2746
- /** Multi-provider env mapping: model prefix → keyEnv (for CLIs like OpenCode that resolve provider from model string) */
2747
- providerEnvMap?: Record<string, {
2748
- keyEnv: string;
2749
- }>;
2750
- /** Env var for inline config (e.g., OPENCODE_CONFIG_CONTENT) — used in gateway mode to set provider base URLs */
2751
- gatewayConfigEnv?: string;
2752
- /** Gateway-only model aliases for CLIs whose native model IDs differ from LiteLLM route names */
2753
- gatewayModelAliases?: Record<string, string>;
2754
- /** Direct-mode model aliases for CLIs whose public model names differ from CLI-native model IDs */
2755
- directModelAliases?: Record<string, string>;
2756
- /** Do not set provider API key env in gateway mode (used when routing via generated settings instead) */
2757
- skipApiKeyEnvInGateway?: boolean;
2758
- /** Dedicated Droid settings file for Evolve gateway custom model routing */
2759
- droidGatewaySettings?: {
2760
- settingsPath: string;
2761
- displayName: string;
2762
- provider: "generic-chat-completion-api" | "openai" | "anthropic";
2763
- maxOutputTokens?: number;
2764
- };
2765
- /** Environment variable that CLI reads for custom outbound HTTP headers */
2766
- customHeadersEnv?: string;
2767
- /** Format for custom headers env var: "newline" (Claude) or "comma" (Gemini). Default: "newline" */
2768
- customHeadersFormat?: "newline" | "comma";
2769
- /**
2770
- * Per-env-var spend tracking for CLIs that support env_http_headers in config
2771
- * (e.g., Codex TOML). Maps LiteLLM header names to env var names that the CLI
2772
- * reads at request time. Alternative to customHeadersEnv for agents without a
2773
- * single custom-headers env var.
2774
- */
2775
- spendTrackingEnvs?: {
2776
- /** Env var name for x-litellm-customer-id value */
2777
- sessionTagEnv: string;
2778
- /** Env var name for x-litellm-tags value */
2779
- runTagEnv: string;
2780
- };
2781
- /**
2782
- * Config-file-based spend tracking for CLIs that read custom headers from a
2783
- * JSON settings file (e.g., Qwen settings.json → model.generationConfig.customHeaders).
2784
- * The SDK writes headers to this file before each run.
2785
- * Source-verified: Qwen reads customHeaders from settings.json, not env vars.
2786
- */
2787
- spendTrackingJsonConfig?: {
2788
- /** JSON path to the customHeaders object (dot-separated) */
2789
- headersPath: string;
2790
- };
2791
- /**
2792
- * TOML provider-based spend tracking for CLIs that read custom_headers from a
2793
- * provider entry in config.toml (e.g., Kimi Code).
2794
- * The SDK writes a provider+model entry with custom_headers before each run.
2795
- * Source-verified: Kimi Code reads custom_headers from
2796
- * providers[name].custom_headers in ~/.kimi-code/config.toml.
2797
- */
2798
- spendTrackingTomlProvider?: {
2799
- /** Config file path (e.g., "~/.kimi-code/config.toml") */
2800
- configPath: string;
2801
- /** Provider name in config (e.g., "evolve-gateway") */
2802
- providerName: string;
2803
- /** Model entry name (e.g., "evolve-default") */
2804
- modelName: string;
2805
- /** Max context size for the model entry */
2806
- maxContextSize: number;
2807
- };
2808
- /** Additional directories to include in checkpoint tar (beyond mcpConfig.settingsDir).
2809
- * Used for agents like OpenCode that spread state across XDG directories. */
2810
- checkpointDirs?: string[];
2811
- /** Additional relative paths to exclude from checkpoint tar. */
2812
- checkpointExcludes?: string[];
3122
+ * Evolve SDK Constants
3123
+ *
3124
+ * Internal constants - not exposed to users.
3125
+ */
3126
+ /**
3127
+ * Sandbox providers the Dashboard runs on the customer's behalf.
3128
+ *
3129
+ * Managed mode means the customer holds one Evolve API key and no provider
3130
+ * credential at all: the Dashboard authenticates the key, records ownership,
3131
+ * and makes the provider call with platform credentials. Each provider has its
3132
+ * own door under /api/managed/<provider>.
3133
+ */
3134
+ declare const MANAGED_SANDBOX_PROVIDERS: readonly ["e2b", "daytona", "modal"];
3135
+ type ManagedSandboxProviderName = (typeof MANAGED_SANDBOX_PROVIDERS)[number];
3136
+
3137
+ /**
3138
+ * Sandbox Provider Resolution
3139
+ *
3140
+ * Resolves default sandbox provider from environment.
3141
+ * Supports E2B, Daytona, and Modal providers.
3142
+ */
3143
+
3144
+ /**
3145
+ * Sandbox-shape defaults a managed provider applies to every sandbox it
3146
+ * creates. Per-create options (`.withSandboxCreateOptions()`) still win, and
3147
+ * the values ride the same validated path as any create: providers and doors
3148
+ * reject what they cannot enforce, never silently ignore it.
3149
+ */
3150
+ interface ManagedSandboxCreateDefaults {
3151
+ /** Lifetime cap (ms) for every sandbox this provider creates. */
3152
+ timeoutMs?: number;
3153
+ /** Compute sizing (cpu cores, memory/disk GiB) for every create. */
3154
+ resources?: SandboxCreateOptions["resources"];
3155
+ }
3156
+ /** Options bag for `managedSandbox()`: the Evolve key plus create defaults. */
3157
+ interface ManagedSandboxOptions extends ManagedSandboxCreateDefaults {
3158
+ /** Evolve API key. Default: the EVOLVE_API_KEY environment variable. */
3159
+ apiKey?: string;
2813
3160
  }
2814
3161
  /**
2815
- * Registry of all supported agents.
3162
+ * A sandbox the platform runs for you.
2816
3163
  *
2817
- * Each agent defines a buildCommand function that constructs the CLI command.
2818
- * This is type-safe and handles conditional logic cleanly.
3164
+ * ```ts
3165
+ * const kit = new Evolve()
3166
+ * .withAgent({ agentType: "claude" })
3167
+ * .withSandbox(await managedSandbox("daytona", { timeoutMs: 7_200_000 }));
3168
+ * ```
3169
+ *
3170
+ * Requires an Evolve API key and no provider credential of any kind. Omit the
3171
+ * provider to take the platform default (E2B). The options bag carries the
3172
+ * Evolve key plus sandbox-shape defaults (`timeoutMs`, `resources`) applied to
3173
+ * every create; a bare string is still accepted as the key alone.
2819
3174
  */
2820
- declare const AGENT_REGISTRY: Record<AgentType, AgentRegistryEntry>;
3175
+ declare function managedSandbox(provider?: ManagedSandboxProviderName, options?: string | ManagedSandboxOptions): Promise<SandboxProvider>;
3176
+
2821
3177
  /**
2822
- * Get registry entry for an agent type
3178
+ * Claude JSONL → ACP-style events parser.
3179
+ *
3180
+ * Native schema source (@anthropic-ai/claude-agent-sdk):
3181
+ * MANUS-API/KNOWLEDGE/claude-agent-sdk/cc_sdk_typescript.md
3182
+ * (SDKMessage, SDKAssistantMessage, SDKPartialAssistantMessage, Tool Input/Output types)
3183
+ *
3184
+ * Conversion logic reference:
3185
+ * MANUS-API/KNOWLEDGE/claude-code-acp/src/tools.ts
3186
+ * (toolInfoFromToolUse, toolUpdateFromToolResult)
3187
+ *
3188
+ * ACP output schema:
3189
+ * MANUS-API/KNOWLEDGE/acp-typescript-sdk/src/schema/types.gen.ts
2823
3190
  */
2824
- declare function getAgentConfig(agentType: AgentType): AgentRegistryEntry;
3191
+
2825
3192
  /**
2826
- * Check if an agent type is valid
3193
+ * Create a Claude parser instance with its own isolated cache.
3194
+ * Each Evolve instance should create its own parser for proper isolation.
2827
3195
  */
2828
- declare function isValidAgentType(type: string): type is AgentType;
3196
+ declare function createClaudeParser(): (jsonLine: string) => OutputEvent[] | null;
3197
+
2829
3198
  /**
2830
- * Expand path with ~ to /home/user
3199
+ * Codex JSONL → ACP-style events parser.
3200
+ *
3201
+ * Native schema: codex-rs/exec/src/exec_events.rs
3202
+ * - ThreadEvent: thread.started, turn.started, turn.completed, item.started, item.updated, item.completed
3203
+ * - ThreadItemDetails: AgentMessage, Reasoning, CommandExecution, FileChange, McpToolCall, WebSearch, TodoList, Error
3204
+ *
3205
+ * ACP output: acp-typescript-sdk/src/schema/types.gen.ts
3206
+ * - SessionUpdate: agent_message_chunk, agent_thought_chunk, tool_call, tool_call_update, plan
3207
+ *
3208
+ * Event mapping:
3209
+ * reasoning → agent_thought_chunk (exec_events.rs:134 ReasoningItem { text })
3210
+ * agent_message → agent_message_chunk (exec_events.rs:129 AgentMessageItem { text })
3211
+ * mcp_tool_call → tool_call/update (exec_events.rs:215 McpToolCallItem)
3212
+ * command_execution → tool_call/update (exec_events.rs:151 CommandExecutionItem)
3213
+ * file_change → tool_call (exec_events.rs:176 FileChangeItem)
3214
+ * todo_list → plan (exec_events.rs:245 TodoListItem { items: TodoItem[] })
3215
+ * web_search → tool_call (exec_events.rs:227 WebSearchItem { query })
2831
3216
  */
2832
- declare function expandPath(path: string): string;
3217
+
2833
3218
  /**
2834
- * Get MCP settings path for an agent
3219
+ * Create a Codex parser instance.
2835
3220
  */
2836
- declare function getMcpSettingsPath(agentType: AgentType): string;
3221
+ declare function createCodexParser(): (jsonLine: string) => OutputEvent[] | null;
3222
+
2837
3223
  /**
2838
- * Get MCP settings directory for an agent
3224
+ * Droid exec parser.
3225
+ *
3226
+ * Supports the documented headless `--output-format stream-json` lines and the
3227
+ * raw `stream-jsonrpc` notification envelope used by Droid's low-level SDK.
3228
+ */
3229
+
3230
+ declare function createDroidParser(): (jsonLine: string) => OutputEvent[] | null;
3231
+
3232
+ /**
3233
+ * Gemini JSONL → ACP-style events parser.
3234
+ *
3235
+ * Native schema (gemini --output-format stream-json):
3236
+ * gemini-cli/packages/core/src/output/types.ts
3237
+ *
3238
+ * Gemini events (types.ts:29-36 JsonStreamEventType):
3239
+ * - "init" → types.ts:43-47 InitEvent { session_id, model }
3240
+ * - "message" → types.ts:49-54 MessageEvent { role, content, delta? }
3241
+ * - "tool_use" → types.ts:56-61 ToolUseEvent { tool_name, tool_id, parameters }
3242
+ * - "tool_result" → types.ts:63-72 ToolResultEvent { tool_id, status, output?, error? }
3243
+ * - "error" → types.ts:74-78 ErrorEvent { severity, message }
3244
+ * - "result" → types.ts:91-99 ResultEvent { status, error?, stats? }
3245
+ *
3246
+ * ACP output: acp-typescript-sdk/src/schema/types.gen.ts:2449-2464
3247
+ */
3248
+
3249
+ /**
3250
+ * Create a Gemini parser instance.
3251
+ */
3252
+ declare function createGeminiParser(): (jsonLine: string) => OutputEvent[] | null;
3253
+
3254
+ /**
3255
+ * Qwen NDJSON → ACP-style events parser.
3256
+ *
3257
+ * Native schema: KNOWLEDGE/qwen-code/packages/sdk-typescript/src/types/protocol.ts
3258
+ * ACP schema: KNOWLEDGE/acp-typescript-sdk/src/schema/types.gen.ts
3259
+ *
3260
+ * Qwen NDJSON message types (protocol.ts:428-433):
3261
+ * - type: "assistant" → SDKAssistantMessage (protocol.ts:102-108)
3262
+ * - type: "stream_event" → SDKPartialAssistantMessage (protocol.ts:225-231)
3263
+ * - type: "user" → SDKUserMessage (protocol.ts:93-100)
3264
+ * - type: "system" → SDKSystemMessage (skipped)
3265
+ * - type: "result" → SDKResultMessage (skipped)
3266
+ *
3267
+ * ContentBlock types (protocol.ts:72-76):
3268
+ * - TextBlock (protocol.ts:43-48): { type: 'text', text: string }
3269
+ * - ThinkingBlock (protocol.ts:49-54): { type: 'thinking', thinking: string }
3270
+ * - ToolUseBlock (protocol.ts:56-62): { type: 'tool_use', id, name, input }
3271
+ * - ToolResultBlock (protocol.ts:64-70): { type: 'tool_result', tool_use_id, content?, is_error? }
3272
+ *
3273
+ * StreamEvent types (protocol.ts:218-223):
3274
+ * - message_start, content_block_start, content_block_delta, content_block_stop, message_stop
3275
+ */
3276
+
3277
+ /**
3278
+ * Stateless parser function (creates new parser per call).
3279
+ * Use createQwenParser() for stateful streaming parsing.
3280
+ *
3281
+ * @param line - Single line of NDJSON from qwen CLI
3282
+ * @returns Array of OutputEvent objects, or null if line couldn't be parsed
3283
+ */
3284
+ declare function parseQwenOutput(line: string): OutputEvent[] | null;
3285
+
3286
+ /**
3287
+ * Unified Parser Entry Point
3288
+ *
3289
+ * Routes NDJSON lines to the appropriate agent-specific parser.
3290
+ * Simple line-based parsing - no buffering needed since CLIs output complete JSON per line.
3291
+ */
3292
+
3293
+ /** Parser function type */
3294
+ type AgentParser = (jsonLine: string) => OutputEvent[] | null;
3295
+ /**
3296
+ * Create a parser instance for the given agent type.
3297
+ * Each Evolve instance should create its own parser for proper isolation.
3298
+ *
3299
+ * @param agentType - The agent type to create a parser for
3300
+ * @returns Parser function that takes NDJSON lines and returns OutputEvents
3301
+ */
3302
+ declare function createAgentParser(agentType: AgentType): AgentParser;
3303
+ /**
3304
+ * Parse a single NDJSON line from any agent (creates new parser per call - use createAgentParser for efficiency)
3305
+ *
3306
+ * @param agentType - The agent type to parse for
3307
+ * @param line - Single line of NDJSON output
3308
+ * @returns Array of OutputEvent objects, or null if line couldn't be parsed
3309
+ */
3310
+ declare function parseNdjsonLine(agentType: AgentType, line: string): OutputEvent[] | null;
3311
+ /**
3312
+ * Parse multiple NDJSON lines (convenience wrapper)
3313
+ *
3314
+ * @param agentType - The agent type to parse for
3315
+ * @param output - Multi-line NDJSON output
3316
+ * @returns Array of all parsed OutputEvent objects
2839
3317
  */
2840
- declare function getMcpSettingsDir(agentType: AgentType): string;
3318
+ declare function parseNdjsonOutput(agentType: AgentType, output: string): OutputEvent[];
2841
3319
 
2842
3320
  /**
2843
3321
  * MCP JSON Configuration Writer
@@ -2859,11 +3337,11 @@ declare function getMcpSettingsDir(agentType: AgentType): string;
2859
3337
  * 1. ${workingDir}/.mcp.json - project-level MCP servers
2860
3338
  * 2. ~/.claude/settings.json - enable project MCP servers
2861
3339
  */
2862
- declare function writeClaudeMcpConfig(sandbox: SandboxInstance, workingDir: string, servers: Record<string, McpServerConfig>): Promise<void>;
3340
+ declare function writeClaudeMcpConfig(sandbox: SandboxInstance, workingDir: string, servers: Record<string, McpServerConfig>, homeDir?: string): Promise<void>;
2863
3341
  /** Write MCP config for Gemini agent */
2864
- declare function writeGeminiMcpConfig(sandbox: SandboxInstance, servers: Record<string, McpServerConfig>): Promise<void>;
3342
+ declare function writeGeminiMcpConfig(sandbox: SandboxInstance, servers: Record<string, McpServerConfig>, homeDir?: string): Promise<void>;
2865
3343
  /** Write MCP config for Qwen agent */
2866
- declare function writeQwenMcpConfig(sandbox: SandboxInstance, servers: Record<string, McpServerConfig>): Promise<void>;
3344
+ declare function writeQwenMcpConfig(sandbox: SandboxInstance, servers: Record<string, McpServerConfig>, homeDir?: string): Promise<void>;
2867
3345
  /**
2868
3346
  * Write MCP config for Droid agent
2869
3347
  *
@@ -2886,13 +3364,18 @@ interface DroidGatewaySettingsConfig {
2886
3364
  * The command passes this file with `droid --settings`, so it does not alter the
2887
3365
  * user's normal ~/.factory/settings.json inside the sandbox.
2888
3366
  */
2889
- declare function writeDroidGatewaySettings(sandbox: SandboxInstance, config: DroidGatewaySettingsConfig, headers: Record<string, string>): Promise<void>;
3367
+ declare function writeDroidGatewaySettings(sandbox: SandboxInstance, config: DroidGatewaySettingsConfig, headers: Record<string, string>, homeDir?: string): Promise<void>;
2890
3368
 
2891
3369
  /**
2892
3370
  * MCP TOML Configuration Writer
2893
3371
  *
2894
3372
  * Handles MCP config for Codex agent which uses TOML format.
2895
3373
  * Uses registry for paths - no hardcoded values.
3374
+ *
3375
+ * Every writer here follows one pattern: parse the existing document into an
3376
+ * object, mutate the object, re-serialize with smol-toml. Nothing inspects the
3377
+ * raw text, so a commented-out section never masks a real one and a value
3378
+ * containing newlines or quotes cannot break the file.
2896
3379
  */
2897
3380
 
2898
3381
  /**
@@ -2901,7 +3384,7 @@ declare function writeDroidGatewaySettings(sandbox: SandboxInstance, config: Dro
2901
3384
  * Codex stores MCP config in ~/.codex/config.toml using TOML format.
2902
3385
  * Format: [mcp_servers.server_name] sections
2903
3386
  */
2904
- declare function writeCodexMcpConfig(sandbox: SandboxInstance, servers: Record<string, McpServerConfig>): Promise<void>;
3387
+ declare function writeCodexMcpConfig(sandbox: SandboxInstance, servers: Record<string, McpServerConfig>, homeDir?: string): Promise<void>;
2905
3388
 
2906
3389
  /**
2907
3390
  * MCP Configuration Module
@@ -2921,7 +3404,7 @@ declare function writeCodexMcpConfig(sandbox: SandboxInstance, servers: Record<s
2921
3404
  * - Droid: JSON to ${workingDir}/.factory/mcp.json
2922
3405
  * - OpenCode: JSON to ${workingDir}/opencode.json (mcp key)
2923
3406
  */
2924
- declare function writeMcpConfig(agentType: AgentType, sandbox: SandboxInstance, workingDir: string, servers: Record<string, McpServerConfig>): Promise<void>;
3407
+ declare function writeMcpConfig(agentType: AgentType, sandbox: SandboxInstance, workingDir: string, servers: Record<string, McpServerConfig>, homeDir?: string): Promise<void>;
2925
3408
 
2926
3409
  /**
2927
3410
  * Prompt Templates
@@ -2991,6 +3474,12 @@ declare const VERIFY_PROMPT: string;
2991
3474
  declare const RETRY_FEEDBACK_PROMPT: string;
2992
3475
  /**
2993
3476
  * Apply template variables to a prompt
3477
+ *
3478
+ * ONE pass over the template: substituted values are emitted verbatim and
3479
+ * never rescanned, so a value that itself contains `{{...}}` (user prompts,
3480
+ * verifier feedback, criteria) is not re-expanded by a later variable — and
3481
+ * `$`-replacement patterns in values stay literal (a function replacement's
3482
+ * return value is never $-interpreted).
2994
3483
  */
2995
3484
  declare function applyTemplate(template: string, variables: Record<string, string>): string;
2996
3485
  /**
@@ -3061,16 +3550,37 @@ declare function readLocalDir(localPath: string, recursive?: boolean): FileMap;
3061
3550
  * const output = await agent.getOutputFiles(true);
3062
3551
  * saveLocalDir('./output', output.files);
3063
3552
  * // Creates: ./output/file.txt, ./output/subdir/nested.txt, etc.
3553
+ *
3554
+ * @throws If an entry's path escapes the target directory. The names come
3555
+ * from sandbox output, so a hostile `../` or absolute entry must not write
3556
+ * outside the directory the caller chose.
3064
3557
  */
3065
3558
  declare function saveLocalDir(localPath: string, files: FileMap): void;
3066
3559
 
3560
+ /**
3561
+ * Configuration Utilities
3562
+ */
3563
+
3564
+ /**
3565
+ * A configuration field the caller got wrong, reported by name.
3566
+ *
3567
+ * Without a check at the door the mistake travels: an empty model or a missing
3568
+ * prompt reaches the command builder and comes back as
3569
+ * `Cannot read properties of undefined (reading 'replace')` thrown by a
3570
+ * shell-quoting helper, which names nothing the caller wrote and points at a
3571
+ * file they have never opened.
3572
+ */
3573
+ declare class EvolveConfigError extends Error {
3574
+ /** The configuration field at fault, e.g. "model" or "prompt". */
3575
+ readonly field: string;
3576
+ constructor(field: string, message: string);
3577
+ }
3578
+
3067
3579
  /**
3068
3580
  * Storage & Checkpointing Module
3069
3581
  *
3070
3582
  * Provides durable persistence for agent workspaces beyond sandbox lifetime.
3071
3583
  * Supports BYOK (user's S3 bucket) and Gateway (Evolve-managed) modes.
3072
- *
3073
- * Evidence: storage-checkpointing plan v2.2
3074
3584
  */
3075
3585
 
3076
3586
  /**
@@ -3189,4 +3699,240 @@ interface SessionsClient {
3189
3699
  */
3190
3700
  declare function sessions(config?: SessionsConfig): SessionsClient;
3191
3701
 
3192
- export { AGENT_REGISTRY, AGENT_TYPES, type ActionbookBrowserConfig, Agent, type AgentBrowserConfig, type AgentConfig, type AgentOptions, type AgentOverride, type AgentParser, type AgentPluginConfig, type AgentRegistryEntry, type AgentResponse, type AgentRuntimeState, type AgentType, BROWSER_ACTIONBOOK_PROMPT, BROWSER_LOGIN_MCP_SERVER_NAME, type BaseMeta, type BestOfConfig, type BestOfParams, type BestOfResult, type BrowserConfig, type BrowserCredentialCreateInput, type BrowserCredentialDeleteInput, type BrowserCredentialListOptions, type BrowserCredentialMetadata, type BrowserCredentialScopeEntry, BrowserCredentialsClient, type BrowserCredentialsClientConfig, type BrowserCredentialsConfig, type BrowserProfileDeleteInput, type BrowserProfileMetadata, BrowserProfilesClient, type BrowserProfilesClientConfig, type BrowserProvider, type BrowserReplay, type BrowserReplayOptions, type BrowserRuntimeInfo, type CandidateCompleteEvent, type CheckpointInfo, type CodexAgentPluginConfig, type DefaultBrowserConfig, type DownloadCheckpointOptions, type DownloadFilesOptions, type DownloadSessionOptions, type EmitOption, type EventHandler, type EventName, Evolve, type EvolveConfig, type EvolveEvents, type ExecuteCommandOptions, type FileMap, type FilterConfig, type FilterParams, type GeminiAgentPluginConfig, type GetEventsOptions, type IndexedMeta, type IntegrationAccount, type IntegrationAccountDeleteParams, type IntegrationAccountDeleteResult, type IntegrationAccountListParams, type IntegrationAccountUpdateParams, type IntegrationAccountUpdateResult, type IntegrationAuthParams, type IntegrationAuthResult, type IntegrationToolsFilter, type IntegrationsConfig, type IntegrationsSetup, type ItemInput, type ItemRetryEvent, JUDGE_PROMPT, type JsonSchema, type JudgeCompleteEvent, type JudgeDecision, type JudgeMeta, type LifecycleEvent, type LifecycleReason, type ListSessionsOptions, type ManagedBrowserProvider, type MapConfig, type MapParams, type MarketplaceAgentPluginConfig, type McpConfigInfo, type McpServerConfig, type ModelInfo, type OnCandidateCompleteCallback, type OnItemRetryCallback, type OnJudgeCompleteCallback, type OnVerifierCompleteCallback, type OnWorkerCompleteCallback, type OperationType, type OutputEvent, type OutputResult, Pipeline, type PipelineContext, type PipelineEventMap, type PipelineEvents, type PipelineResult, type ProcessInfo, type Prompt, type PromptFn, RETRY_FEEDBACK_PROMPT, type ReasoningEffort, type ReduceConfig, type ReduceMeta, type ReduceParams, type ReduceResult, type ResolvedStorageConfig, type RetryConfig, type RunCost, type RunOptions, SCHEMA_PROMPT, SWARM_RESULT_BRAND, SYSTEM_PROMPT, type SandboxCommandHandle, type SandboxCommandResult, type SandboxCommands, type SandboxCreateOptions, type SandboxFiles, type SandboxInstance, type SandboxLifecycleState, type SandboxProvider, type SandboxRunOptions, type SandboxSpawnOptions, type SchemaValidationOptions, Semaphore, type SessionCost, type SessionEvent, type SessionInfo, type SessionPage, type SessionStatus, type SessionsClient, type SessionsConfig, type SkillName, type SkillsConfig, type StepCompleteEvent, type StepErrorEvent, type StepEvent, type StepResult, type StepStartEvent, type StorageClient, type StorageConfig, type StreamCallbacks, Swarm, type SwarmConfig, type SwarmResult, SwarmResultList, TerminalPipeline, VALIDATION_PRESETS, VERIFY_PROMPT, type ValidationMode, type VerifierCompleteEvent, type VerifyConfig, type VerifyDecision, type VerifyInfo, type VerifyMeta, WORKSPACE_PROMPT, WORKSPACE_SWE_PROMPT, type WorkerCompleteEvent, type WorkspaceMode, applyTemplate, browserCredentials, browserProfiles, buildWorkerSystemPrompt, createAgentParser, createClaudeParser, createCodexParser, createDroidParser, createGeminiParser, executeWithRetry, expandPath, getAgentConfig, getMcpSettingsDir, getMcpSettingsPath, isValidAgentType, isZodSchema, jsonSchemaToString, parseNdjsonLine, parseNdjsonOutput, parseQwenOutput, readLocalDir, resolveStorageConfig, saveLocalDir, sessions, storage, writeClaudeMcpConfig, writeCodexMcpConfig, writeDroidGatewaySettings, writeDroidMcpConfig, writeGeminiMcpConfig, writeMcpConfig, writeQwenMcpConfig, zodSchemaToJson };
3702
+ /**
3703
+ * A typed failure from the hosted evals API.
3704
+ *
3705
+ * `message` is the server's own product sentence and `code` is the stable
3706
+ * machine-readable identifier, so callers branch on codes and never on English.
3707
+ * `code` is typed as the closed HostedErrorCode union (widened to string for
3708
+ * forward compatibility with a newer server), which is what makes a typo like
3709
+ * `insufficient_creidts` a compile error instead of a branch that never runs.
3710
+ *
3711
+ * `param` and `details` are the machine-readable half of the refusal:
3712
+ *
3713
+ * catch (err) {
3714
+ * if (err instanceof EvolveApiError && err.code === "provider_unsupported") {
3715
+ * // every refused task WITH its reason — not a sentence to regex
3716
+ * const refused = err.details?.refused_tasks as { task_name: string }[];
3717
+ * }
3718
+ * }
3719
+ *
3720
+ * The server truncates the MESSAGE when a list is long and never truncates
3721
+ * `details`, so the data is always complete even when the sentence says
3722
+ * "and 8 more".
3723
+ */
3724
+ declare class EvolveApiError extends Error {
3725
+ /** HTTP status of the failed response */
3726
+ readonly status: number;
3727
+ /** Stable snake_case error code from the API ("unknown_error" when absent) */
3728
+ readonly code: HostedErrorCode | "unknown_error" | (string & {});
3729
+ /**
3730
+ * The input field this refusal is about — a body path ("agents[0].name"),
3731
+ * a query parameter ("limit"), or a multipart part name ("run_command").
3732
+ * Undefined when the failure is not about a particular field.
3733
+ */
3734
+ readonly param?: string;
3735
+ /** The complete machine-readable data behind the message. Never truncated. */
3736
+ readonly details?: Record<string, unknown>;
3737
+ /**
3738
+ * Seconds to wait before retrying (429/503). Read from the body first and the
3739
+ * Retry-After header second, because a browser fetch cannot always see the
3740
+ * header on a cross-origin response.
3741
+ */
3742
+ readonly retryAfterSec?: number;
3743
+ /** Server-side id for this failure; the string to quote in a support thread. */
3744
+ readonly requestId?: string;
3745
+ constructor(status: number, code: string, message: string, extra?: {
3746
+ param?: string;
3747
+ details?: Record<string, unknown>;
3748
+ retryAfterSec?: number;
3749
+ requestId?: string;
3750
+ });
3751
+ /** True when this code is one this SDK version knows about. */
3752
+ isKnownCode(): boolean;
3753
+ }
3754
+ /**
3755
+ * Thrown by datasets().getActive() when the named dataset exists but has no
3756
+ * active version, so there is no runnable version to resolve. Use get() to
3757
+ * inspect a dataset that may not have an active version yet.
3758
+ */
3759
+ declare class NoActiveVersionError extends Error {
3760
+ /** The dataset name that had no active version */
3761
+ readonly dataset: string;
3762
+ constructor(dataset: string);
3763
+ }
3764
+ /**
3765
+ * Downloaded bytes did not match the digest the server stated for them.
3766
+ *
3767
+ * NOT an EvolveApiError: the request succeeded, so it gets its own type rather
3768
+ * than an invented error code.
3769
+ */
3770
+ declare class EvolveDigestMismatchError extends Error {
3771
+ readonly expected: string;
3772
+ readonly actual: string;
3773
+ readonly name = "EvolveDigestMismatchError";
3774
+ constructor(expected: string, actual: string);
3775
+ }
3776
+ /**
3777
+ * A download ended early: fewer bytes arrived than Content-Length promised.
3778
+ *
3779
+ * Its own type because a truncated body is not a wrong body — the distinction
3780
+ * tells a caller whether to retry (yes) or to stop trusting the stored object
3781
+ * (that is the digest error).
3782
+ */
3783
+ declare class EvolveIncompleteDownloadError extends Error {
3784
+ readonly expectedBytes: number;
3785
+ readonly receivedBytes: number;
3786
+ readonly name = "EvolveIncompleteDownloadError";
3787
+ constructor(expectedBytes: number, receivedBytes: number);
3788
+ }
3789
+ /**
3790
+ * Create a DatasetsClient for the shared dataset catalog.
3791
+ *
3792
+ * Requires EVOLVE_API_KEY (or { apiKey } in config).
3793
+ *
3794
+ * @example
3795
+ * ```ts
3796
+ * import { datasets } from "@evolvingmachines/sdk";
3797
+ *
3798
+ * const d = datasets();
3799
+ * const catalog = await d.list();
3800
+ * const deepSwe = await d.get("deep-swe@1.1");
3801
+ * ```
3802
+ */
3803
+ declare function datasets(config?: HostedClientConfig): DatasetsClient;
3804
+ /**
3805
+ * Create an AgentsClient for the caller's own private registered agents.
3806
+ *
3807
+ * Register an agent once, then name it in job `agents[].name` exactly
3808
+ * like a built-in. Requires EVOLVE_API_KEY (or { apiKey } in config).
3809
+ *
3810
+ * @example
3811
+ * ```ts
3812
+ * import { agents, jobs } from "@evolvingmachines/sdk";
3813
+ *
3814
+ * const registered = agents();
3815
+ * await registered.create({
3816
+ * name: "acme-cli",
3817
+ * install_script: "curl -fsSL https://acme.dev/install.sh | sh",
3818
+ * run_command: "acme-cli --headless",
3819
+ * });
3820
+ *
3821
+ * await jobs().start({
3822
+ * datasets: [{ name: "deep-swe" }],
3823
+ * agents: [{ name: "acme-cli", model_name: "gpt-5.5" }],
3824
+ * max_trial_spend_usd: 25,
3825
+ * });
3826
+ * ```
3827
+ */
3828
+ declare function agents(config?: HostedClientConfig): AgentsClient;
3829
+ /**
3830
+ * Create a JobsClient for hosted jobs.
3831
+ *
3832
+ * Requires EVOLVE_API_KEY (or { apiKey } in config).
3833
+ *
3834
+ * @example
3835
+ * ```ts
3836
+ * import { jobs } from "@evolvingmachines/sdk";
3837
+ *
3838
+ * const client = jobs();
3839
+ * // datasets: bare name = active version; { name, version } pins one
3840
+ * const job = await client.start({
3841
+ * datasets: [{ name: "deep-swe" }],
3842
+ * agents: [{ name: "codex", model_name: "gpt-5.5" }],
3843
+ * n_attempts: 1,
3844
+ * n_concurrent_trials: 4,
3845
+ * max_trial_spend_usd: 25,
3846
+ * });
3847
+ * const final = await client.watch(job.id, {
3848
+ * onEvent: (event) => console.log(event.type, event.data),
3849
+ * });
3850
+ * ```
3851
+ */
3852
+ declare function jobs(config?: HostedClientConfig): JobsClient;
3853
+ /**
3854
+ * Create a TrialsClient. A trial id is globally addressable — no method here
3855
+ * takes a job id; the trial body carries `job_id` as the reverse pointer.
3856
+ *
3857
+ * Requires EVOLVE_API_KEY (or { apiKey } in config).
3858
+ */
3859
+ declare function trials(config?: HostedClientConfig): TrialsClient;
3860
+ /**
3861
+ * Create an AuthClient for caller identity.
3862
+ *
3863
+ * Requires EVOLVE_API_KEY (or { apiKey } in config).
3864
+ */
3865
+ declare function auth(config?: HostedClientConfig): AuthClient;
3866
+ /**
3867
+ * The hosted surface, configured once.
3868
+ *
3869
+ * The four factories are the right decomposition — a dataset catalog, your own
3870
+ * agent registrations, jobs, and globally addressable trials are genuinely
3871
+ * different lifetimes — but they made you say the same thing four times:
3872
+ *
3873
+ * const d = datasets({ apiKey, baseUrl });
3874
+ * const a = agents({ apiKey, baseUrl }); // again
3875
+ * const j = jobs({ apiKey, baseUrl }); // and again
3876
+ *
3877
+ * and any one of those going out of sync with the others is a bug that looks
3878
+ * like a permissions problem. One door, one config:
3879
+ *
3880
+ * const evolve = hosted({ apiKey });
3881
+ * const catalog = await evolve.datasets.list();
3882
+ * const job = await evolve.jobs.start({ ... });
3883
+ *
3884
+ * The clients are built LAZILY, on first access. That matters because they
3885
+ * throw when no API key is present, and `meta()` needs no key at all — so
3886
+ * `hosted().meta()` works on a signed-out page, while `hosted().jobs` still
3887
+ * fails loudly and immediately the moment you reach for something that does
3888
+ * need credentials.
3889
+ */
3890
+ interface HostedEvolve {
3891
+ /** The dataset catalog: list, get, publish, download, delete. */
3892
+ readonly datasets: DatasetsClient;
3893
+ /** Your own bring-your-own agent registrations. */
3894
+ readonly agents: AgentsClient;
3895
+ /** Jobs: start, watch, compare, resume, regrade, download. */
3896
+ readonly jobs: JobsClient;
3897
+ /** Globally addressable trials: get, trace, artifact, regrade, stop. */
3898
+ readonly trials: TrialsClient;
3899
+ /**
3900
+ * The capability document — every agent, provider, status, limit, and
3901
+ * error code the platform supports. Public: no API key required.
3902
+ *
3903
+ * Fetch it once and stop hardcoding. It is what tells you the legal agent
3904
+ * names without having to send a bad one and read the 400.
3905
+ */
3906
+ meta(): Promise<CapabilityDocument>;
3907
+ }
3908
+ /**
3909
+ * Open the hosted surface with one configuration.
3910
+ *
3911
+ * Named `hosted()` rather than `evolve()` deliberately: `Evolve` is already the
3912
+ * local-sandbox SDK class in this same package, and two exports one shift key
3913
+ * apart that do completely different things is a trap. `hosted()` says which
3914
+ * half of the SDK you are reaching for.
3915
+ *
3916
+ * @example
3917
+ * ```ts
3918
+ * import { hosted } from "@evolvingmachines/sdk";
3919
+ *
3920
+ * const evolve = hosted(); // EVOLVE_API_KEY from env
3921
+ * const { agents } = await evolve.meta(); // no key needed for this one
3922
+ * const job = await evolve.jobs.start({
3923
+ * datasets: [{ name: "deep-swe" }],
3924
+ * agents: [{ name: "claude", model_name: "claude-fable-5" }],
3925
+ * });
3926
+ * ```
3927
+ */
3928
+ declare function hosted(config?: HostedClientConfig): HostedEvolve;
3929
+ /**
3930
+ * Fetch the capability document.
3931
+ *
3932
+ * NO API KEY. The document is the same information the docs publish, and
3933
+ * requiring credentials would mean a signed-out page could not populate its own
3934
+ * agent picker — so this is the one hosted call that takes only a base URL.
3935
+ */
3936
+ declare function meta(config?: HostedClientConfig): Promise<CapabilityDocument>;
3937
+
3938
+ export { AGENT_REGISTRY, AGENT_TYPES, type ActionbookBrowserConfig, Agent, type AgentBrowserConfig, type AgentConfig, type AgentError, type AgentOptions, type AgentOverride, type AgentParser, type AgentPluginConfig, type AgentRegistryEntry, type AgentResponse, type AgentRuntimeState, type AgentType, AgentsClient, AuthClient, BINARY_EFFORT_VALUES, BROWSER_ACTIONBOOK_PROMPT, BROWSER_LOGIN_MCP_SERVER_NAME, type BaseMeta, type BestOfConfig, type BestOfParams, type BestOfResult, type BrowserConfig, type BrowserCredentialCreateInput, type BrowserCredentialDeleteInput, type BrowserCredentialListOptions, type BrowserCredentialMetadata, type BrowserCredentialScopeEntry, BrowserCredentialsClient, type BrowserCredentialsClientConfig, type BrowserCredentialsConfig, type BrowserProfileDeleteInput, type BrowserProfileMetadata, BrowserProfilesClient, type BrowserProfilesClientConfig, type BrowserProvider, type BrowserReplay, type BrowserReplayOptions, type BrowserRuntimeInfo, type CandidateCompleteEvent, CapabilityDocument, type CheckpointInfo, type CodexAgentPluginConfig, DEFAULT_REASONING_EFFORT, DatasetsClient, type DefaultBrowserConfig, type DownloadCheckpointOptions, type DownloadFilesOptions, type DownloadSessionOptions, type EffortSupport, type EmitOption, type EventHandler, type EventName, Evolve, EvolveApiError, type EvolveConfig, EvolveConfigError, EvolveDigestMismatchError, type EvolveEvents, EvolveIncompleteDownloadError, type ExecuteCommandOptions, type ExternalGatewayConfig, type FileMap, type FilterConfig, type FilterParams, type GeminiAgentPluginConfig, type GetEventsOptions, HostedClientConfig, HostedErrorCode, type HostedEvolve, type IndexedMeta, type IntegrationAccount, type IntegrationAccountDeleteParams, type IntegrationAccountDeleteResult, type IntegrationAccountListParams, type IntegrationAccountUpdateParams, type IntegrationAccountUpdateResult, type IntegrationAuthParams, type IntegrationAuthResult, type IntegrationToolsFilter, type IntegrationsConfig, type IntegrationsSetup, type ItemInput, type ItemRetryEvent, JUDGE_PROMPT, JobsClient, type JsonSchema, type JudgeCompleteEvent, type JudgeDecision, type JudgeMeta, type LifecycleEvent, type LifecycleReason, type ListSessionsOptions, MANAGED_SANDBOX_PROVIDERS, type ManagedBrowserProvider, type ManagedSandboxCreateDefaults, type ManagedSandboxOptions, type ManagedSandboxProviderName, type ManagedSecretMetadata, type ManagedSecretRef, type ManagedSecretsClient, type ManagedSecretsClientConfig, type MapConfig, type MapParams, type MarketplaceAgentPluginConfig, type McpConfigInfo, type McpServerConfig, type ModelInfo, NoActiveVersionError, type OnCandidateCompleteCallback, type OnItemRetryCallback, type OnJudgeCompleteCallback, type OnVerifierCompleteCallback, type OnWorkerCompleteCallback, type OperationType, type OutputEvent, type OutputResult, Pipeline, type PipelineContext, type PipelineEventMap, type PipelineEvents, type PipelineResult, type ProcessInfo, type Prompt, type PromptFn, REASONING_EFFORTS, RETRY_FEEDBACK_PROMPT, type ReasoningEffort, type ReduceConfig, type ReduceMeta, type ReduceParams, type ReduceResult, type ResolvedStorageConfig, type RetryConfig, type RunCost, type RunOptions, SCHEMA_PROMPT, SWARM_RESULT_BRAND, SYSTEM_PROMPT, type SandboxCommandHandle, type SandboxCommandResult, type SandboxCommands, type SandboxCreateOptions, type SandboxFiles, type SandboxInstance, type SandboxLifecycleState, type SandboxNetworkPolicy, type SandboxProvider, type SandboxRunOptions, type SandboxSpawnOptions, type SchemaValidationOptions, Semaphore, type SessionCost, type SessionEvent, type SessionInfo, type SessionPage, type SessionStatus, type SessionUpdate, type SessionsClient, type SessionsConfig, type SkillName, type SkillsConfig, type StepCompleteEvent, type StepErrorEvent, type StepEvent, type StepResult, type StepStartEvent, type StorageClient, type StorageConfig, type StreamCallbacks, Swarm, type SwarmConfig, type SwarmResult, SwarmResultList, TerminalPipeline, TrialsClient, VALIDATION_PRESETS, VERIFY_PROMPT, type ValidationMode, type VerifierCompleteEvent, type VerifyConfig, type VerifyDecision, type VerifyInfo, type VerifyMeta, WORKSPACE_PROMPT, WORKSPACE_SWE_PROMPT, type WorkerCompleteEvent, type WorkspaceMode, agents, applyTemplate, auth, browserCredentials, browserProfiles, buildWorkerSystemPrompt, createAgentParser, createClaudeParser, createCodexParser, createDroidParser, createGeminiParser, datasets, executeWithRetry, expandPath, getAgentConfig, getMcpSettingsDir, getMcpSettingsPath, harnessEffortVocabulary, hosted, isAgentWorkUpdate, isValidAgentType, isZodSchema, jobs, jsonSchemaToString, managedSandbox, managedSecrets, meta, parseNdjsonLine, parseNdjsonOutput, parseQwenOutput, readLocalDir, resolveStorageConfig, saveLocalDir, sessions, storage, trials, writeClaudeMcpConfig, writeCodexMcpConfig, writeDroidGatewaySettings, writeDroidMcpConfig, writeGeminiMcpConfig, writeMcpConfig, writeQwenMcpConfig, zodSchemaToJson };