@ai-sdk/anthropic 4.0.57 → 4.0.59

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -385,6 +385,57 @@ const servedByFallback = iterations?.some(
385
385
  console.log('Served by fallback:', servedByFallback);
386
386
  ```
387
387
 
388
+ ### Dangerous Tool Use Safeguard
389
+
390
+ The `safeguards` option asks the API to run additional server-side checks as part of the request. The `dangerous_tool_use` safeguard classifies every `tool_use` block in the response for dangerous actions, such as destructive shell commands or data exfiltration, and returns a verdict per tool call. This is the check that Claude Code's auto mode relies on. The `dangerous-tool-use-2026-09-03` beta header is added for you.
391
+
392
+ ```ts highlight="9-16"
393
+ import { anthropic, AnthropicLanguageModelOptions } from '@ai-sdk/anthropic';
394
+ import { generateText, tool } from 'ai';
395
+ import { z } from 'zod';
396
+
397
+ const result = await generateText({
398
+ model: anthropic('claude-sonnet-4-5'),
399
+ prompt: 'Use the bash tool to run: echo hello',
400
+ tools: { bash: tool({ inputSchema: z.object({ command: z.string() }) }) },
401
+ providerOptions: {
402
+ anthropic: {
403
+ safeguards: [
404
+ {
405
+ type: 'dangerous_tool_use',
406
+ classifierContext: { v: 1, permission_mode: 'auto' },
407
+ },
408
+ ],
409
+ } satisfies AnthropicLanguageModelOptions,
410
+ },
411
+ });
412
+
413
+ console.log(result.providerMetadata?.anthropic?.safeguardResults);
414
+ ```
415
+
416
+ The verdicts are available on `providerMetadata.anthropic.safeguardResults`, one entry per requested safeguard, in the API's wire shape:
417
+
418
+ ```json
419
+ [
420
+ {
421
+ "type": "dangerous_tool_use",
422
+ "status": {
423
+ "type": "available",
424
+ "tool_uses": {
425
+ "toolu_01Ti4QpUfhLV6QqCTFVY8C7w": {
426
+ "type": "evaluated",
427
+ "outcome": "not_flagged"
428
+ }
429
+ }
430
+ }
431
+ }
432
+ ]
433
+ ```
434
+
435
+ `status.type` is `available` when the classifier ran; `unsupported` means the API key is not enabled for the beta. Inside `status.tool_uses`, each tool call id maps to `evaluated` (with an `outcome` of `not_flagged` or `flagged`, the latter with an `explanation` such as `[Data Exfiltration]`), `skipped`, or `unavailable` (a transient classifier failure). When streaming, the verdicts arrive on the final `message_delta` event and are exposed on the `finish` part's provider metadata.
436
+
437
+ Anthropic's platform reference does not document the beta yet; the feature is described from the client side in the Claude Code [auto mode classifier](https://code.claude.com/docs/en/auto-mode-classifier-billing) and [gateway compatibility](https://code.claude.com/docs/en/llm-gateway-protocol#feature-pass-through) guides.
438
+
388
439
  ### Reasoning
389
440
 
390
441
  Anthropic models support extended thinking, where Claude shows its reasoning process before providing a final answer.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ai-sdk/anthropic",
3
- "version": "4.0.57",
3
+ "version": "4.0.59",
4
4
  "type": "module",
5
5
  "license": "Apache-2.0",
6
6
  "sideEffects": false,
@@ -36,7 +36,7 @@
36
36
  },
37
37
  "dependencies": {
38
38
  "@ai-sdk/provider": "4.0.17",
39
- "@ai-sdk/provider-utils": "5.0.44"
39
+ "@ai-sdk/provider-utils": "5.0.45"
40
40
  },
41
41
  "devDependencies": {
42
42
  "@ai-sdk/test-server": "2.0.1",
@@ -657,6 +657,27 @@ const anthropicStopDetailsSchema = z.object({
657
657
 
658
658
  export type AnthropicStopDetails = z.infer<typeof anthropicStopDetailsSchema>;
659
659
 
660
+ const anthropicSafeguardResultSchema = z.object({
661
+ type: z.string(),
662
+ status: z.object({
663
+ type: z.string(),
664
+ tool_uses: z
665
+ .record(
666
+ z.string(),
667
+ z.object({
668
+ type: z.string(),
669
+ outcome: z.string().nullish(),
670
+ explanation: z.string().nullish(),
671
+ }),
672
+ )
673
+ .nullish(),
674
+ }),
675
+ });
676
+
677
+ export type AnthropicSafeguardResult = z.infer<
678
+ typeof anthropicSafeguardResultSchema
679
+ >;
680
+
660
681
  const anthropicToolCallCallerSchema = z.union([
661
682
  z.object({
662
683
  type: z.literal('code_execution_20250825'),
@@ -1004,6 +1025,7 @@ export const anthropicResponseSchema = lazySchema(() =>
1004
1025
  input_transformations: z
1005
1026
  .array(anthropicInputTransformationSchema)
1006
1027
  .nullish(),
1028
+ safeguard_results: z.array(anthropicSafeguardResultSchema).nullish(),
1007
1029
  usage: z.looseObject({
1008
1030
  input_tokens: z.number(),
1009
1031
  output_tokens: z.number(),
@@ -1412,6 +1434,7 @@ export const anthropicChunkSchema = lazySchema(() =>
1412
1434
  stop_reason: z.string().nullish(),
1413
1435
  stop_sequence: z.string().nullish(),
1414
1436
  stop_details: anthropicStopDetailsSchema.nullish(),
1437
+ safeguard_results: z.array(anthropicSafeguardResultSchema).nullish(),
1415
1438
  container: z
1416
1439
  .object({
1417
1440
  expires_at: z.string(),
@@ -1098,6 +1098,9 @@ function convertAnthropicMessageMetadata(response: AnthropicResponse) {
1098
1098
  ...(response.input_transformations != null
1099
1099
  ? { inputTransformations: response.input_transformations }
1100
1100
  : {}),
1101
+ ...(response.safeguard_results != null
1102
+ ? { safeguardResults: response.safeguard_results }
1103
+ : {}),
1101
1104
  iterations: response.usage.iterations
1102
1105
  ? response.usage.iterations.map(
1103
1106
  iteration =>
@@ -355,6 +355,25 @@ export const anthropicLanguageModelOptions = z.object({
355
355
  */
356
356
  anthropicBeta: z.array(z.string()).optional(),
357
357
 
358
+ /**
359
+ * Server-side safeguards to run as part of the request.
360
+ *
361
+ * `dangerous_tool_use` asks the API to classify every `tool_use` block in
362
+ * the response for dangerous actions (the check Claude Code's auto mode
363
+ * relies on). The per-call verdicts are returned in
364
+ * `providerMetadata.anthropic.safeguardResults`, keyed by tool call id.
365
+ * `classifierContext` is passed through to the API as `classifier_context`.
366
+ * The `dangerous-tool-use-2026-09-03` beta is added automatically.
367
+ */
368
+ safeguards: z
369
+ .array(
370
+ z.object({
371
+ type: z.literal('dangerous_tool_use'),
372
+ classifierContext: z.record(z.string(), z.unknown()).optional(),
373
+ }),
374
+ )
375
+ .optional(),
376
+
358
377
  contextManagement: z
359
378
  .object({
360
379
  edits: z.array(
@@ -671,6 +671,16 @@ export class AnthropicLanguageModel implements LanguageModelV4 {
671
671
  system: messagesPrompt.system,
672
672
  messages: messagesPrompt.messages,
673
673
 
674
+ ...(anthropicOptions?.safeguards &&
675
+ anthropicOptions.safeguards.length > 0 && {
676
+ safeguards: anthropicOptions.safeguards.map(safeguard => ({
677
+ type: safeguard.type,
678
+ ...(safeguard.classifierContext !== undefined && {
679
+ classifier_context: safeguard.classifierContext,
680
+ }),
681
+ })),
682
+ }),
683
+
674
684
  ...(contextManagement && {
675
685
  context_management: {
676
686
  edits: contextManagement.edits
@@ -810,6 +820,13 @@ export class AnthropicLanguageModel implements LanguageModelV4 {
810
820
  betas.add('mcp-client-2025-04-04');
811
821
  }
812
822
 
823
+ if (
824
+ anthropicOptions?.safeguards &&
825
+ anthropicOptions.safeguards.length > 0
826
+ ) {
827
+ betas.add('dangerous-tool-use-2026-09-03');
828
+ }
829
+
813
830
  if (contextManagement) {
814
831
  betas.add('context-management-2025-06-27');
815
832
 
@@ -1560,6 +1577,9 @@ export class AnthropicLanguageModel implements LanguageModelV4 {
1560
1577
  ...(response.input_transformations != null
1561
1578
  ? { inputTransformations: response.input_transformations }
1562
1579
  : {}),
1580
+ ...(response.safeguard_results != null
1581
+ ? { safeguardResults: response.safeguard_results }
1582
+ : {}),
1563
1583
 
1564
1584
  iterations: response.usage.iterations
1565
1585
  ? response.usage.iterations.map(
@@ -1697,6 +1717,7 @@ export class AnthropicLanguageModel implements LanguageModelV4 {
1697
1717
  let stopSequence: string | null = null;
1698
1718
  let stopDetails: AnthropicMessageMetadata['stopDetails'] = undefined;
1699
1719
  let inputTransformations: AnthropicMessageMetadata['inputTransformations'];
1720
+ let safeguardResults: AnthropicMessageMetadata['safeguardResults'];
1700
1721
  let container: AnthropicMessageMetadata['container'] | null = null;
1701
1722
  let isJsonResponseFromTool = false;
1702
1723
  let isMessageOpen = false;
@@ -2701,6 +2722,12 @@ export class AnthropicLanguageModel implements LanguageModelV4 {
2701
2722
  inputTransformations = value.input_transformations;
2702
2723
  }
2703
2724
 
2725
+ // Earlier deltas may carry null while the classifier is still
2726
+ // running; the last non-null value is the final verdict.
2727
+ if (value.delta.safeguard_results != null) {
2728
+ safeguardResults = value.delta.safeguard_results;
2729
+ }
2730
+
2704
2731
  rawUsage = {
2705
2732
  ...rawUsage,
2706
2733
  ...(value.usage as JSONObject),
@@ -2720,6 +2747,7 @@ export class AnthropicLanguageModel implements LanguageModelV4 {
2720
2747
  ...(inputTransformations != null
2721
2748
  ? { inputTransformations }
2722
2749
  : {}),
2750
+ ...(safeguardResults != null ? { safeguardResults } : {}),
2723
2751
  iterations: usage.iterations
2724
2752
  ? usage.iterations.map(
2725
2753
  iter =>
@@ -1,4 +1,5 @@
1
1
  import type { JSONObject } from '@ai-sdk/provider';
2
+ import type { AnthropicSafeguardResult } from './anthropic-api';
2
3
 
3
4
  /**
4
5
  * Represents a single iteration in the usage breakdown.
@@ -62,6 +63,19 @@ export interface AnthropicMessageMetadata {
62
63
  reason: string;
63
64
  }>;
64
65
 
66
+ /**
67
+ * Verdicts of the server-side safeguards requested with
68
+ * `providerOptions.anthropic.safeguards`, one entry per requested safeguard.
69
+ * For `dangerous_tool_use`, `status.type` is `'available'` when the
70
+ * classifier ran and `status.tool_uses` maps each tool call id to its
71
+ * verdict (`type: 'evaluated'` with an `outcome` of `'not_flagged'` or
72
+ * `'flagged'` plus an `explanation`, `'skipped'`, or `'unavailable'`).
73
+ *
74
+ * Kept in the API's wire shape (snake_case), since consumers such as Claude
75
+ * Code read the verdicts verbatim.
76
+ */
77
+ safeguardResults?: AnthropicSafeguardResult[];
78
+
65
79
  /**
66
80
  * Details about why the request stopped. Present only when the API returns
67
81
  * a `refusal` stop reason together with a `stop_details` object (a