@ai-sdk/harness 1.0.132 → 1.0.133

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md CHANGED
@@ -1,5 +1,15 @@
1
1
  # @ai-sdk/harness
2
2
 
3
+ ## 1.0.133
4
+
5
+ ### Patch Changes
6
+
7
+ - b2049fa: fix(harness): expose `resolveSandboxCredentialEnvironment()` helper in favor of `createSandboxCredentialEnvironment()` and use it across bridge backed adapters
8
+ - 11f0e71: feat (harness): add `HarnessAgentSession.readHistory({ since })` contract to return normalized session history, to implement per individual harness adapter
9
+ - 2b9195b: fix(harness): emit tool lifecycle events and results as each tool runs
10
+ - 446725d: fix(harness): surface detach failures and keep session handles usable
11
+ - ai@7.0.122
12
+
3
13
  ## 1.0.132
4
14
 
5
15
  ### Patch Changes
@@ -3,7 +3,7 @@ import { Experimental_SandboxSession, UserModelMessage, ToolResultPart, Provider
3
3
  import { ToolsContextSettings } from 'ai/internal';
4
4
  import { OutputInterface, AgentCallParameters, Prompt, StopCondition, GenerateTextOnStartCallback, GenerateTextOnStepStartCallback, OnLanguageModelCallStartCallback, OnLanguageModelCallEndCallback, OnToolExecutionStartCallback, OnToolExecutionEndCallback, GenerateTextOnStepEndCallback, GenerateTextOnEndCallback, ToolApprovalStatus, TelemetryOptions, ActiveTools, StreamTextResult, Agent, GenerateTextResult, AgentStreamParameters, Telemetry } from 'ai';
5
5
  import { z } from 'zod/v4';
6
- import { JSONSchema7, JSONValue, LanguageModelV4StreamPart, LanguageModelV4ToolCall, LanguageModelV4ToolApprovalRequest, LanguageModelV4ToolResult, LanguageModelV4FinishReason, LanguageModelV4Usage, AISDKError } from '@ai-sdk/provider';
6
+ import { JSONSchema7, JSONValue, LanguageModelV4StreamPart, LanguageModelV4ToolCall, LanguageModelV4ToolApprovalRequest, LanguageModelV4ToolResult, LanguageModelV4FinishReason, LanguageModelV4Usage, LanguageModelV4TextPart, LanguageModelV4FilePart, LanguageModelV4CustomPart, LanguageModelV4ReasoningPart, LanguageModelV4ReasoningFilePart, LanguageModelV4ToolCallPart, LanguageModelV4ToolResultPart, LanguageModelV4ToolResultOutput, LanguageModelV4ToolApprovalResponsePart, AISDKError } from '@ai-sdk/provider';
7
7
 
8
8
  /**
9
9
  * One file to write into the sandbox as part of an adapter's bootstrap recipe.
@@ -681,6 +681,65 @@ type HarnessV1StreamPart = {
681
681
  rawValue: unknown;
682
682
  };
683
683
 
684
+ type LanguageModelV4ToolResultContentPart = Extract<LanguageModelV4ToolResultOutput, {
685
+ type: 'content';
686
+ }>['value'][number];
687
+ type HarnessV1TextPart = Omit<LanguageModelV4TextPart, 'providerOptions'>;
688
+ type HarnessV1FilePart = Omit<LanguageModelV4FilePart, 'providerOptions'>;
689
+ type HarnessV1CustomPart = Omit<LanguageModelV4CustomPart, 'providerOptions'>;
690
+ type HarnessV1ReasoningPart = Omit<LanguageModelV4ReasoningPart, 'providerOptions'>;
691
+ type HarnessV1ReasoningFilePart = Omit<LanguageModelV4ReasoningFilePart, 'providerOptions'>;
692
+ type HarnessV1ToolCallPart = Omit<LanguageModelV4ToolCallPart, 'providerOptions'> & {
693
+ nativeName?: string;
694
+ };
695
+ type HarnessV1ToolResultOutput = Omit<Extract<LanguageModelV4ToolResultOutput, {
696
+ type: 'text';
697
+ }>, 'providerOptions'> | Omit<Extract<LanguageModelV4ToolResultOutput, {
698
+ type: 'json';
699
+ }>, 'providerOptions'> | Omit<Extract<LanguageModelV4ToolResultOutput, {
700
+ type: 'execution-denied';
701
+ }>, 'providerOptions'> | Omit<Extract<LanguageModelV4ToolResultOutput, {
702
+ type: 'error-text';
703
+ }>, 'providerOptions'> | Omit<Extract<LanguageModelV4ToolResultOutput, {
704
+ type: 'error-json';
705
+ }>, 'providerOptions'> | {
706
+ type: 'content';
707
+ value: Array<Omit<Extract<LanguageModelV4ToolResultContentPart, {
708
+ type: 'text';
709
+ }>, 'providerOptions'> | Omit<Extract<LanguageModelV4ToolResultContentPart, {
710
+ type: 'file';
711
+ }>, 'providerOptions'> | Omit<Extract<LanguageModelV4ToolResultContentPart, {
712
+ type: 'custom';
713
+ }>, 'providerOptions'>>;
714
+ };
715
+ type HarnessV1ToolResultPart = Omit<LanguageModelV4ToolResultPart, 'output' | 'providerOptions'> & {
716
+ output: HarnessV1ToolResultOutput;
717
+ };
718
+ type HarnessV1ToolApprovalResponsePart = Omit<LanguageModelV4ToolApprovalResponsePart, 'providerOptions'>;
719
+ type HarnessV1UserMessage = {
720
+ readonly role: 'user';
721
+ readonly content: Array<HarnessV1TextPart | HarnessV1FilePart>;
722
+ readonly at?: string;
723
+ readonly harnessMetadata?: HarnessV1Metadata;
724
+ };
725
+ type HarnessV1AssistantMessage = {
726
+ readonly role: 'assistant';
727
+ readonly content: Array<HarnessV1TextPart | HarnessV1FilePart | HarnessV1CustomPart | HarnessV1ReasoningPart | HarnessV1ReasoningFilePart | HarnessV1ToolCallPart | HarnessV1ToolResultPart>;
728
+ readonly at?: string;
729
+ readonly harnessMetadata?: HarnessV1Metadata;
730
+ };
731
+ type HarnessV1ToolMessage = {
732
+ readonly role: 'tool';
733
+ readonly content: Array<HarnessV1ToolResultPart | HarnessV1ToolApprovalResponsePart>;
734
+ readonly at?: string;
735
+ readonly harnessMetadata?: HarnessV1Metadata;
736
+ };
737
+ /**
738
+ * A persisted harness message using V4 prompt content shapes and metadata
739
+ * scoped to the adapter that produced it.
740
+ */
741
+ type HarnessV1Message = HarnessV1UserMessage | HarnessV1AssistantMessage | HarnessV1ToolMessage;
742
+
684
743
  type HarnessV1BuiltinToolFiltering = {
685
744
  mode: 'allow';
686
745
  toolNames: string[];
@@ -757,6 +816,14 @@ type HarnessV1StartOptions = {
757
816
  */
758
817
  readonly sessionWorkDir: string;
759
818
  };
819
+ /**
820
+ * Result of `HarnessV1Session.doReadHistory`.
821
+ */
822
+ type HarnessV1ReadHistoryResult = {
823
+ readonly messages: ReadonlyArray<HarnessV1Message>;
824
+ /** Opaque position; pass back as `since` to read only what follows. */
825
+ readonly cursor: string;
826
+ };
760
827
  /**
761
828
  * Options passed to `HarnessV1Session.doPromptTurn`.
762
829
  */
@@ -846,6 +913,33 @@ type HarnessV1Session = {
846
913
  * `customInstructions`, when supported, steer the compaction summary.
847
914
  */
848
915
  doCompact(customInstructions?: string): PromiseLike<void>;
916
+ /**
917
+ * Read the conversation history the runtime itself persisted, normalized
918
+ * to `HarnessV1Message`.
919
+ *
920
+ * The session's history can grow outside the harness contract: the same
921
+ * runtime conversation may be continued interactively (`claude --resume`),
922
+ * by another process, or before this session attached. Hosts that render a
923
+ * continuous record of the conversation — not just the turns they drove —
924
+ * need to read that history back, and the runtime's own store is the only
925
+ * source that has it. The adapter owns its runtime's persistence format,
926
+ * so the read belongs here rather than in every host.
927
+ *
928
+ * `since` is the `cursor` from a previous read; the result then contains
929
+ * only messages recorded after it. The cursor is adapter-owned and opaque
930
+ * to the host.
931
+ *
932
+ * Optional capability: adapters that cannot read their runtime's store
933
+ * omit the method entirely, and the agent surfaces that as
934
+ * `HarnessCapabilityUnsupportedError`. An adapter that implements it but
935
+ * cannot reach the store from the current environment (e.g. it lives
936
+ * inside a remote sandbox) throws `HarnessHistoryUnavailableError`. A
937
+ * conversation with no recorded messages yet is not an error — it resolves
938
+ * to an empty `messages` array.
939
+ */
940
+ doReadHistory?(options: {
941
+ readonly since?: string;
942
+ }): PromiseLike<HarnessV1ReadHistoryResult>;
849
943
  /**
850
944
  * Continue the in-flight turn **without a new user prompt**, returning the
851
945
  * same control surface as `doPromptTurn`. Used to keep consuming a turn that
@@ -1639,6 +1733,24 @@ declare class HarnessAgentSession {
1639
1733
  * unusable.
1640
1734
  */
1641
1735
  detach(): Promise<HarnessAgentResumeSessionState>;
1736
+ /**
1737
+ * Read the conversation history the runtime itself persisted, normalized
1738
+ * by the adapter. Includes exchanges that happened outside this process —
1739
+ * the same conversation continued interactively in the agent's own CLI,
1740
+ * for instance — which the live event stream never saw.
1741
+ *
1742
+ * Pass a previous result's `cursor` as `since` to read only the delta. A
1743
+ * conversation with no recorded messages yet resolves to an empty
1744
+ * `messages` array.
1745
+ *
1746
+ * Throws `HarnessCapabilityUnsupportedError` when the adapter does not
1747
+ * implement history reads, and
1748
+ * `HarnessHistoryUnavailableError` when the adapter supports them but
1749
+ * cannot reach the runtime's store from this environment.
1750
+ */
1751
+ readHistory(options?: {
1752
+ since?: string;
1753
+ }): Promise<HarnessV1ReadHistoryResult>;
1642
1754
  /**
1643
1755
  * Persist enough state to resume later, then stop the runtime and any
1644
1756
  * harness-owned sandbox.
@@ -1753,17 +1753,25 @@ function runPrompt(input) {
1753
1753
  let stepApprovalRequests = [];
1754
1754
  let bufferedToolOutcomes = [];
1755
1755
  const toolExecutions = /* @__PURE__ */ new Map();
1756
+ const publishToolExecutionStart = async (execution) => {
1757
+ execution.startNotification ??= lifecycle.toolExecutionStart({
1758
+ toolCall: execution.toolCall
1759
+ });
1760
+ await execution.startNotification;
1761
+ };
1762
+ const publishToolExecutionEnd = async (execution) => {
1763
+ if (execution.toolOutput == null) return;
1764
+ await publishToolExecutionStart(execution);
1765
+ execution.endNotification ??= lifecycle.toolExecutionEnd({
1766
+ toolCall: execution.toolCall,
1767
+ toolOutput: execution.toolOutput,
1768
+ toolExecutionMs: execution.toolExecutionMs ?? 0
1769
+ });
1770
+ await execution.endNotification;
1771
+ };
1756
1772
  const publishToolExecutions = async () => {
1757
1773
  for (const execution of toolExecutions.values()) {
1758
- if (execution.toolOutput == null) continue;
1759
- await lifecycle.toolExecutionStart({
1760
- toolCall: execution.toolCall
1761
- });
1762
- await lifecycle.toolExecutionEnd({
1763
- toolCall: execution.toolCall,
1764
- toolOutput: execution.toolOutput,
1765
- toolExecutionMs: execution.toolExecutionMs ?? 0
1766
- });
1774
+ await publishToolExecutionEnd(execution);
1767
1775
  }
1768
1776
  for (const publish of bufferedToolOutcomes) publish();
1769
1777
  toolExecutions.clear();
@@ -2048,6 +2056,9 @@ function runPrompt(input) {
2048
2056
  toolExecutions.set(rawToolCall.toolCallId, {
2049
2057
  toolCall
2050
2058
  });
2059
+ await publishToolExecutionStart(
2060
+ toolExecutions.get(rawToolCall.toolCallId)
2061
+ );
2051
2062
  const executionStartedAt = Date.now();
2052
2063
  const execution = await maybeExecuteHostTool({
2053
2064
  event: rawToolCall,
@@ -2068,15 +2079,13 @@ function runPrompt(input) {
2068
2079
  },
2069
2080
  input.sessionWorkDir
2070
2081
  );
2071
- bufferedToolOutcomes.push(() => {
2072
- result.enqueue({
2073
- type: "tool-result",
2074
- toolCallId: rawToolCall.toolCallId,
2075
- toolName: rawToolCall.toolName,
2076
- input: void 0,
2077
- output: stripped.result,
2078
- preliminary: true
2079
- });
2082
+ result.enqueue({
2083
+ type: "tool-result",
2084
+ toolCallId: rawToolCall.toolCallId,
2085
+ toolName: rawToolCall.toolName,
2086
+ input: void 0,
2087
+ output: stripped.result,
2088
+ preliminary: true
2080
2089
  });
2081
2090
  }
2082
2091
  });
@@ -2091,13 +2100,11 @@ function runPrompt(input) {
2091
2100
  outcome: execution.outcome
2092
2101
  });
2093
2102
  toolExecution.toolExecutionMs = Date.now() - executionStartedAt;
2094
- bufferedToolOutcomes.push(() => {
2095
- enqueueHostToolOutcome({
2096
- toolCall,
2097
- outcome: execution.outcome
2098
- });
2103
+ await publishToolExecutionEnd(toolExecution);
2104
+ enqueueHostToolOutcome({
2105
+ toolCall,
2106
+ outcome: execution.outcome
2099
2107
  });
2100
- await publishToolExecutions();
2101
2108
  return "continued";
2102
2109
  };
2103
2110
  try {
@@ -2239,11 +2246,7 @@ function runPrompt(input) {
2239
2246
  displayValue,
2240
2247
  translateOptions
2241
2248
  );
2242
- if (value.type === "tool-result") {
2243
- bufferedToolOutcomes.push(() => {
2244
- for (const part of translatedParts) result.enqueue(part);
2245
- });
2246
- } else {
2249
+ if (value.type !== "tool-result") {
2247
2250
  for (const part of translatedParts) result.enqueue(part);
2248
2251
  }
2249
2252
  let validatedHostToolCall;
@@ -2277,9 +2280,13 @@ function runPrompt(input) {
2277
2280
  const toolCall = toolCallsByToolCallId.get(value.toolCallId);
2278
2281
  if (toolCall != null) {
2279
2282
  stepToolCalls.push(toolCall);
2280
- toolExecutions.set(value.toolCallId, {
2283
+ const execution = {
2281
2284
  toolCall
2282
- });
2285
+ };
2286
+ toolExecutions.set(value.toolCallId, execution);
2287
+ if (value.providerExecuted === true) {
2288
+ await publishToolExecutionStart(execution);
2289
+ }
2283
2290
  }
2284
2291
  }
2285
2292
  if (value.type === "tool-result") {
@@ -2295,11 +2302,18 @@ function runPrompt(input) {
2295
2302
  output: value.result
2296
2303
  };
2297
2304
  }
2305
+ if (execution?.toolExecutionMs == null && execution?.executionStartedAt != null) {
2306
+ execution.toolExecutionMs = Date.now() - execution.executionStartedAt;
2307
+ }
2298
2308
  if (rawToolCallsByToolCallId.get(value.toolCallId)?.providerExecuted === true && execution?.toolOutput != null) {
2299
2309
  stepProviderToolResults.push(
2300
2310
  execution.toolOutput
2301
2311
  );
2302
2312
  }
2313
+ if (execution != null) {
2314
+ await publishToolExecutionEnd(execution);
2315
+ }
2316
+ for (const part of translatedParts) result.enqueue(part);
2303
2317
  }
2304
2318
  if (value.type === "tool-approval-request") {
2305
2319
  const toolCall = toolCallsByToolCallId.get(value.toolCallId);
@@ -2457,16 +2471,20 @@ function runPrompt(input) {
2457
2471
  type: "execution-denied",
2458
2472
  reason: customToolApprovalDecision.reason
2459
2473
  };
2474
+ const execution = toolExecutions.get(toolCall.toolCallId);
2475
+ if (execution != null) {
2476
+ await publishToolExecutionStart(execution);
2477
+ }
2460
2478
  await submitToolResult({
2461
2479
  toolCallId: toolCall.toolCallId,
2462
2480
  output
2463
2481
  });
2464
- const execution = toolExecutions.get(toolCall.toolCallId);
2465
2482
  if (execution != null) {
2466
2483
  execution.toolOutput = toToolOutput({
2467
2484
  toolCall: parsedToolCall,
2468
2485
  outcome: { ok: true, output }
2469
2486
  });
2487
+ await publishToolExecutionEnd(execution);
2470
2488
  }
2471
2489
  continue;
2472
2490
  }
@@ -2531,7 +2549,15 @@ function runPrompt(input) {
2531
2549
  }
2532
2550
  startHostToolExecution(
2533
2551
  (async () => {
2552
+ const toolExecution = toolExecutions.get(toolCall.toolCallId);
2553
+ if (toolExecution == null) {
2554
+ throw new Error(
2555
+ `Harness '${input.harness.harnessId}' could not track host tool '${toolCall.toolName}'.`
2556
+ );
2557
+ }
2558
+ await publishToolExecutionStart(toolExecution);
2534
2559
  const executionStartedAt = Date.now();
2560
+ toolExecution.executionStartedAt = executionStartedAt;
2535
2561
  const execution = await maybeExecuteHostTool({
2536
2562
  event: toolCall,
2537
2563
  parsedToolCall: validatedHostToolCall,
@@ -2551,15 +2577,13 @@ function runPrompt(input) {
2551
2577
  },
2552
2578
  input.sessionWorkDir
2553
2579
  );
2554
- bufferedToolOutcomes.push(() => {
2555
- result.enqueue({
2556
- type: "tool-result",
2557
- toolCallId: toolCall.toolCallId,
2558
- toolName: toolCall.toolName,
2559
- input: void 0,
2560
- output: stripped.result,
2561
- preliminary: true
2562
- });
2580
+ result.enqueue({
2581
+ type: "tool-result",
2582
+ toolCallId: toolCall.toolCallId,
2583
+ toolName: toolCall.toolName,
2584
+ input: void 0,
2585
+ output: stripped.result,
2586
+ preliminary: true
2563
2587
  });
2564
2588
  }
2565
2589
  });
@@ -2568,14 +2592,12 @@ function runPrompt(input) {
2568
2592
  `Harness '${input.harness.harnessId}' could not execute host tool '${toolCall.toolName}'.`
2569
2593
  );
2570
2594
  }
2571
- const toolExecution = toolExecutions.get(toolCall.toolCallId);
2572
- if (toolExecution != null) {
2573
- toolExecution.toolOutput = toToolOutput({
2574
- toolCall: parsedToolCall,
2575
- outcome: execution.outcome
2576
- });
2577
- toolExecution.toolExecutionMs = Date.now() - executionStartedAt;
2578
- }
2595
+ toolExecution.toolOutput = toToolOutput({
2596
+ toolCall: parsedToolCall,
2597
+ outcome: execution.outcome
2598
+ });
2599
+ toolExecution.toolExecutionMs = Date.now() - executionStartedAt;
2600
+ await publishToolExecutionEnd(toolExecution);
2579
2601
  })()
2580
2602
  );
2581
2603
  }
@@ -3023,22 +3045,47 @@ var HarnessAgentSession = class {
3023
3045
  );
3024
3046
  }
3025
3047
  const session = this.underlyingSession;
3026
- try {
3027
- if (this.turnState !== "idle") {
3028
- return this.toResumeStateWithContinuation({
3029
- continueFrom: await this.finalizeCurrentTurnSuspension({ session })
3030
- });
3031
- }
3032
- const raw = await session.doDetach();
3033
- const validated = await validateLifecycleStateData({
3034
- harness: this.harness,
3035
- state: raw,
3036
- expectedType: "resume-session"
3048
+ const state = this.turnState !== "idle" ? this.toResumeStateWithContinuation({
3049
+ continueFrom: await this.finalizeCurrentTurnSuspension({ session })
3050
+ }) : await validateLifecycleStateData({
3051
+ harness: this.harness,
3052
+ state: await session.doDetach(),
3053
+ expectedType: "resume-session"
3054
+ });
3055
+ this.endLocalHandle({ sessionState: "detached" });
3056
+ return state;
3057
+ }
3058
+ /**
3059
+ * Read the conversation history the runtime itself persisted, normalized
3060
+ * by the adapter. Includes exchanges that happened outside this process —
3061
+ * the same conversation continued interactively in the agent's own CLI,
3062
+ * for instance — which the live event stream never saw.
3063
+ *
3064
+ * Pass a previous result's `cursor` as `since` to read only the delta. A
3065
+ * conversation with no recorded messages yet resolves to an empty
3066
+ * `messages` array.
3067
+ *
3068
+ * Throws `HarnessCapabilityUnsupportedError` when the adapter does not
3069
+ * implement history reads, and
3070
+ * `HarnessHistoryUnavailableError` when the adapter supports them but
3071
+ * cannot reach the runtime's store from this environment.
3072
+ */
3073
+ async readHistory(options) {
3074
+ if (this.sessionState !== "active" || this.underlyingSession == null) {
3075
+ throw new Error(
3076
+ `Harness session ${this.sessionId} is not active and cannot read history.`
3077
+ );
3078
+ }
3079
+ const session = this.underlyingSession;
3080
+ if (typeof session.doReadHistory !== "function") {
3081
+ throw new HarnessCapabilityUnsupportedError({
3082
+ harnessId: this.harness.harnessId,
3083
+ message: `Harness '${this.harness.harnessId}' does not support reading the runtime's conversation history.`
3037
3084
  });
3038
- return validated;
3039
- } finally {
3040
- this.endLocalHandle({ sessionState: "detached" });
3041
3085
  }
3086
+ return await session.doReadHistory({
3087
+ ...options?.since != null ? { since: options.since } : {}
3088
+ });
3042
3089
  }
3043
3090
  /**
3044
3091
  * Persist enough state to resume later, then stop the runtime and any