claude-agent-sdk 1.1.0 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (46) hide show
  1. checksums.yaml +4 -4
  2. data/.yardopts +10 -0
  3. data/CHANGELOG.md +90 -0
  4. data/README.md +43 -31
  5. data/docs/cli-installer.md +26 -4
  6. data/docs/client.md +29 -11
  7. data/docs/configuration.md +164 -1
  8. data/docs/errors.md +32 -2
  9. data/docs/hooks-and-permissions.md +30 -10
  10. data/docs/mcp-servers.md +30 -9
  11. data/docs/observability.md +61 -10
  12. data/docs/options.md +232 -0
  13. data/docs/rails.md +263 -18
  14. data/docs/sessions.md +40 -12
  15. data/docs/subagents.md +1 -1
  16. data/docs/types.md +100 -11
  17. data/lib/claude_agent_sdk/cli_installer.rb +140 -19
  18. data/lib/claude_agent_sdk/command_builder.rb +84 -27
  19. data/lib/claude_agent_sdk/fiber_boundary.rb +45 -2
  20. data/lib/claude_agent_sdk/instrumentation/otel.rb +90 -28
  21. data/lib/claude_agent_sdk/query.rb +228 -77
  22. data/lib/claude_agent_sdk/railtie.rb +27 -2
  23. data/lib/claude_agent_sdk/sdk_mcp_server.rb +78 -26
  24. data/lib/claude_agent_sdk/session_mutations.rb +112 -92
  25. data/lib/claude_agent_sdk/session_resume.rb +356 -39
  26. data/lib/claude_agent_sdk/session_store.rb +31 -2
  27. data/lib/claude_agent_sdk/sessions.rb +720 -138
  28. data/lib/claude_agent_sdk/subprocess_cli_transport.rb +227 -29
  29. data/lib/claude_agent_sdk/testing/session_store_conformance.rb +18 -7
  30. data/lib/claude_agent_sdk/transcript_mirror_batcher.rb +45 -37
  31. data/lib/claude_agent_sdk/transport.rb +28 -12
  32. data/lib/claude_agent_sdk/types/attributes.rb +9 -0
  33. data/lib/claude_agent_sdk/types/base.rb +85 -15
  34. data/lib/claude_agent_sdk/types/hooks.rb +73 -0
  35. data/lib/claude_agent_sdk/types/mcp.rb +37 -1
  36. data/lib/claude_agent_sdk/types/option_values.rb +186 -4
  37. data/lib/claude_agent_sdk/types/options.rb +35 -5
  38. data/lib/claude_agent_sdk/types/permissions.rb +18 -9
  39. data/lib/claude_agent_sdk/version.rb +1 -1
  40. data/lib/claude_agent_sdk.rb +94 -46
  41. data/lib/generators/claude_agent_sdk/install/templates/claude_agent_sdk.rb.tt +6 -0
  42. data/sig/claude_agent_sdk/types/hooks.rbs +6 -3
  43. data/sig/claude_agent_sdk/types/option_values.rbs +23 -6
  44. data/sig/claude_agent_sdk/types/options.rbs +11 -7
  45. data/sig/claude_agent_sdk/types/permissions.rbs +4 -2
  46. metadata +6 -4
@@ -135,6 +135,23 @@ module ClaudeAgentSDK
135
135
  end
136
136
  end
137
137
 
138
+ # What a failing user callback raises — a hook, can_use_tool, an SDK MCP
139
+ # tool / resource / prompt handler, or the callback_wrapper around one.
140
+ # The Ruby spelling of Python's `except Exception`: NotImplementedError
141
+ # and LoadError (both ScriptError), SystemStackError and SecurityError
142
+ # are not StandardErrors, so `rescue StandardError` lets them through.
143
+ # The control request then goes unanswered and its handler task ends
144
+ # with an exception Async treats as fatal for the whole reactor. Rescue
145
+ # `*FiberBoundary::CALLBACK_FAILURES` wherever a callback's failure is
146
+ # turned into the answer the CLI is waiting for.
147
+ #
148
+ # An explicit list, not `rescue Exception`: cancellation (Async::Stop,
149
+ # InlineCancellation) and process exits (SystemExit, SignalException —
150
+ # see .invoke_callback) must keep propagating. NoMemoryError stays out
151
+ # as well: building the answer would most likely fail again.
152
+ # @api private
153
+ CALLBACK_FAILURES = [StandardError, ScriptError, SystemStackError, SecurityError].freeze
154
+
138
155
  # Carries a SystemExit / SignalException (Interrupt included) raised by
139
156
  # a user callback out of the FiberBoundary hop — see .invoke_callback.
140
157
  # A StandardError so the hop ends normally: a :thread worker that died
@@ -335,9 +352,35 @@ module ClaudeAgentSDK
335
352
  Thread.current.report_on_exception = false
336
353
  work.call
337
354
  end
338
- return thread.value if timeout.nil?
339
- raise JoinTimeout, "timed out after #{timeout}s" unless thread.join(timeout)
355
+ await_worker(thread, timeout)
356
+ end
357
+
358
+ # Wait for the worker thread of one hop and return its value (or re-raise
359
+ # what it raised); with +timeout+, raise JoinTimeout once that many
360
+ # seconds have passed.
361
+ #
362
+ # Only "the thread has finished" or "the deadline has really passed" ends
363
+ # the wait. Under a fiber scheduler Thread#join parks the fiber, and MRI
364
+ # reports ANY wakeup that finds the thread alive as a timeout: join
365
+ # returns nil, and Thread#value returns nil with it. The wakeup need not
366
+ # belong to this join — one queued by an earlier hop's thread stays behind
367
+ # when an exception (Async::Stop, a deadline) is raised into the fiber
368
+ # before it is consumed, and resumes this hop instead. Trusting a single
369
+ # join then returned nil while the callback was still running, or raised
370
+ # JoinTimeout with no time passed.
371
+ # @api private
372
+ def await_worker(thread, timeout)
373
+ if timeout.nil?
374
+ nil until thread.join
375
+ return thread.value
376
+ end
340
377
 
378
+ deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + timeout
379
+ until thread.join([deadline - Process.clock_gettime(Process::CLOCK_MONOTONIC), 0].max)
380
+ next if Process.clock_gettime(Process::CLOCK_MONOTONIC) < deadline
381
+
382
+ raise JoinTimeout, "timed out after #{timeout}s"
383
+ end
341
384
  thread.value
342
385
  end
343
386
 
@@ -7,8 +7,10 @@ module ClaudeAgentSDK
7
7
  module Instrumentation
8
8
  # OpenTelemetry observer that emits spans for Claude Agent SDK messages.
9
9
  #
10
- # Uses standard gen_ai.* semantic conventions recognized by Langfuse, Datadog,
11
- # Jaeger, and other OTel-compatible backends.
10
+ # Span attributes follow the Langfuse and OpenInference conventions, plus a
11
+ # subset of the OTel gen_ai.* attributes. docs/observability.md lists every
12
+ # attribute and says where they differ from the OTel GenAI semantic
13
+ # conventions.
12
14
  #
13
15
  # Requires the `opentelemetry-api` gem at runtime. Users must configure
14
16
  # `opentelemetry-sdk` and an exporter (e.g., `opentelemetry-exporter-otlp`)
@@ -49,10 +51,12 @@ module ClaudeAgentSDK
49
51
  @default_attributes = default_attributes
50
52
  @root_span = nil
51
53
  @root_context = nil
54
+ @trace_started = false # an InitMessage opened a trace since this observer was created or last closed
52
55
  @tool_spans = {} # tool_use_id => span
53
56
  @first_user_input = nil # first user prompt of the current trace
54
57
  @pending_prompt = nil # prompt that belongs to the NEXT trace (see on_user_prompt)
55
58
  @last_assistant_text = nil # capture last assistant text for trace output
59
+ @usage_reported_message_ids = Set.new # API responses whose usage a generation span of this trace carries
56
60
  @cost_session_id = nil
57
61
  @last_total_cost_usd = nil
58
62
  end
@@ -99,16 +103,25 @@ module ClaudeAgentSDK
99
103
  end
100
104
  end
101
105
 
102
- # Recording-only by design: a Client session can survive an error (the
103
- # user may rescue one bad message and keep receiving), so finishing here
104
- # would orphan later spans and break the next turn. Finish ownership
105
- # stays with end_trace/on_close; start_trace also finishes any dangling
106
- # span from a previous trace as the never-disconnected backstop.
106
+ # With a trace open this is recording-only by design: a Client session
107
+ # can survive an error (the user may rescue one bad message and keep
108
+ # receiving), so finishing here would orphan later spans and break the
109
+ # next turn. Finish ownership stays with end_trace/on_close; start_trace
110
+ # also finishes any dangling span from a previous trace as the
111
+ # never-disconnected backstop.
112
+ #
113
+ # With no trace open, an error that arrives before the session's first
114
+ # InitMessage gets a session span of its own (record_failed_start). Once
115
+ # a trace has run, an error between traces stays unrecorded: the usual
116
+ # one is the ResultError query() raises when the CLI exits non-zero
117
+ # after an error result, and the trace that just ended already reports
118
+ # that failure.
107
119
  def on_error(error)
108
- return unless @root_span
109
-
110
- @root_span.record_exception(error)
111
- @root_span.status = OpenTelemetry::Trace::Status.error(error.message)
120
+ if @root_span
121
+ record_error(@root_span, error)
122
+ elsif !@trace_started
123
+ record_failed_start(error)
124
+ end
112
125
  end
113
126
 
114
127
  def on_close
@@ -120,11 +133,12 @@ module ClaudeAgentSDK
120
133
  @pending_prompt = nil
121
134
  @cost_session_id = nil
122
135
  @last_total_cost_usd = nil
136
+ @trace_started = false
123
137
  end
124
138
 
125
139
  private
126
140
 
127
- def start_trace(message) # rubocop:disable Metrics/AbcSize -- flat mapping of init-message fields to root span attributes
141
+ def start_trace(message)
128
142
  # A new init without an intervening ResultMessage (e.g. /clear or an
129
143
  # interrupted turn) supersedes the current trace; finish it so it is
130
144
  # exported instead of leaking as a never-ended span, and reset the
@@ -141,21 +155,9 @@ module ClaudeAgentSDK
141
155
  # because it came first chronologically.
142
156
  @first_user_input = @pending_prompt if @pending_prompt
143
157
  @pending_prompt = nil
158
+ @trace_started = true
144
159
 
145
- attrs = {
146
- # gen_ai semantic conventions (recognized by Langfuse, Datadog, etc.)
147
- 'gen_ai.system' => 'anthropic',
148
- 'gen_ai.request.model' => message.model,
149
- # OpenInference conventions (recognized by Langfuse, Arize)
150
- 'openinference.span.kind' => 'AGENT',
151
- 'llm.model_name' => message.model,
152
- 'input.mime_type' => 'text/plain',
153
- 'output.mime_type' => 'text/plain',
154
- # Langfuse: 'agent' type triggers the trace flow diagram (DAG graph)
155
- 'langfuse.observation.type' => 'agent',
156
- # Session tracking
157
- 'session.id' => message.session_id
158
- }.merge(@default_attributes)
160
+ attrs = session_span_attrs(model: message.model, session_id: message.session_id)
159
161
 
160
162
  if message.respond_to?(:claude_code_version) && message.claude_code_version
161
163
  attrs['claude_code.version'] = message.claude_code_version
@@ -174,6 +176,50 @@ module ClaudeAgentSDK
174
176
  @root_span.set_attribute('input.value', truncate(@first_user_input))
175
177
  end
176
178
 
179
+ # The attributes every session span starts with. The model and the
180
+ # session id come from the InitMessage, so the span of a session that
181
+ # failed before sending one (record_failed_start) has neither.
182
+ def session_span_attrs(model: nil, session_id: nil)
183
+ {
184
+ # gen_ai semantic conventions (recognized by Langfuse, Datadog, etc.)
185
+ 'gen_ai.system' => 'anthropic',
186
+ 'gen_ai.request.model' => model,
187
+ # OpenInference conventions (recognized by Langfuse, Arize)
188
+ 'openinference.span.kind' => 'AGENT',
189
+ 'llm.model_name' => model,
190
+ 'input.mime_type' => 'text/plain',
191
+ 'output.mime_type' => 'text/plain',
192
+ # Langfuse: 'agent' type triggers the trace flow diagram (DAG graph)
193
+ 'langfuse.observation.type' => 'agent',
194
+ # Session tracking
195
+ 'session.id' => session_id
196
+ }.merge(@default_attributes)
197
+ end
198
+
199
+ # A session that fails before its first InitMessage (the CLI cannot be
200
+ # found or started, initialize fails or times out, the process dies
201
+ # before its first frame) has no trace to record the error on, so the
202
+ # error gets a session span of its own, with the prompt if one was sent.
203
+ #
204
+ # The span is finished here and never becomes @root_span. query() does
205
+ # call on_close next, but a Client#connect that fails before the
206
+ # handshake never does, and an unfinished span is never exported; with
207
+ # no root span set, a later on_close has nothing to finish a second
208
+ # time. The buffers are reset so that the failed session's prompt cannot
209
+ # label the next trace of a reused observer.
210
+ def record_failed_start(error)
211
+ span = @tracer.start_span('claude_agent.session', attributes: compact_attrs(session_span_attrs))
212
+ span.set_attribute('input.value', truncate(@first_user_input)) if @first_user_input
213
+ record_error(span, error)
214
+ span.finish
215
+ reset_session_buffers
216
+ end
217
+
218
+ def record_error(span, error)
219
+ span.record_exception(error)
220
+ span.status = OpenTelemetry::Trace::Status.error(error.message)
221
+ end
222
+
177
223
  def handle_assistant(message)
178
224
  return unless @root_context
179
225
 
@@ -203,7 +249,7 @@ module ClaudeAgentSDK
203
249
  'gen_ai.completion' => truncate(combined_text),
204
250
  # OpenInference: Langfuse maps output.value to the Preview Output field
205
251
  'output.value' => truncate(combined_text)
206
- }.merge(usage_token_attrs(message.usage || {}))
252
+ }.merge(generation_usage_attrs(message))
207
253
 
208
254
  OpenTelemetry::Context.with_current(@root_context) do
209
255
  span = @tracer.start_span('claude_agent.generation', attributes: compact_attrs(attrs))
@@ -343,10 +389,12 @@ module ClaudeAgentSDK
343
389
 
344
390
  # Clear per-trace buffers so a reused observer instance (sequential
345
391
  # query() calls or multi-turn Client sessions) does not stamp stale
346
- # input/output onto later traces.
392
+ # input/output onto later traces, and so the message ids remembered by
393
+ # generation_usage_attrs never outlive their trace.
347
394
  def reset_session_buffers
348
395
  @first_user_input = nil
349
396
  @last_assistant_text = nil
397
+ @usage_reported_message_ids.clear
350
398
  end
351
399
 
352
400
  def record_retry_event(message)
@@ -397,6 +445,20 @@ module ClaudeAgentSDK
397
445
  }
398
446
  end
399
447
 
448
+ # gen_ai.usage.* attributes for one generation span. The CLI sends one
449
+ # assistant frame per content block, and every frame of an API response
450
+ # repeats that response's message id and usage snapshot, so putting the
451
+ # snapshot on every span counted the response's input, cache-read and
452
+ # cache-creation tokens once per block. Only the first span of a message
453
+ # id carries it. A message without an id cannot be grouped and keeps its
454
+ # usage; a frame without usage does not use up its id.
455
+ def generation_usage_attrs(message)
456
+ attrs = compact_attrs(usage_token_attrs(message.usage || {}))
457
+ return attrs if attrs.empty? || message.message_id.nil?
458
+
459
+ @usage_reported_message_ids.add?(message.message_id) ? attrs : {}
460
+ end
461
+
400
462
  # Usage hashes arrive symbol-keyed from the live CLI (symbolize_names:
401
463
  # true) and string-keyed from session transcripts.
402
464
  def usage_value(usage, key)