claude-agent-sdk 0.33.0 → 0.34.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -19,6 +19,11 @@ module ClaudeAgentSDK
19
19
  CLI_PATH_ENV_VAR = 'CLAUDE_CLI_PATH'
20
20
  VERSION_CHECK_TIMEOUT_SECONDS = 2 # mirrors Python's anyio.fail_after(2)
21
21
  RECENT_STDERR_LINES_LIMIT = 20
22
+ # After stdout EOF the child has closed (or lost) its last stdout handle,
23
+ # so it is normally already exiting; a CLI still running this long
24
+ # afterwards is wedged and gets the same TERM -> KILL ladder as #close.
25
+ EOF_EXIT_GRACE_SECONDS = 5
26
+ EOF_TERM_GRACE_SECONDS = 2
22
27
 
23
28
  # Track live CLI subprocesses so we can terminate them when the parent Ruby
24
29
  # process exits. Mirrors the Python (PR #916, a `set[Process]`) and
@@ -110,6 +115,17 @@ module ClaudeAgentSDK
110
115
  # close can nil @stdin between write's readiness check and the actual
111
116
  # @stdin.write call, producing NoMethodError on nil.
112
117
  @stdin_mutex = Mutex.new
118
+ # Writers holding a live @stdin snapshot (inside #write's IO call),
119
+ # keyed by Fiber.current, valued by that fiber's Fiber.scheduler (nil
120
+ # for a plain thread). Inserted in #write's snapshot critical section,
121
+ # so once stdin is detached under @stdin_mutex the registry holds
122
+ # exactly the writers that can still touch the IO; deletes and reads
123
+ # are single GVL-atomic Hash calls and take no lock (a lock in #write's
124
+ # ensure could suspend a writer mid-unwind). Consulted when stdin is
125
+ # closed: closing the fd neither wakes a fiber parked in IO#write nor
126
+ # lets a reactor-side IO#close return while a plain thread is parked
127
+ # there — see #wake_parked_fiber_writers / #close_stdin_io.
128
+ @inflight_writers = {}
113
129
  end
114
130
 
115
131
  # Probe order (first hit wins):
@@ -152,18 +168,25 @@ module ClaudeAgentSDK
152
168
  end
153
169
  return cli if cli && !cli.empty? && File.executable?(cli)
154
170
 
155
- # Try common locations
171
+ # Try common locations. The home-relative ones are skipped when no
172
+ # usable home exists (see #home_dir), so a HOME-less container still
173
+ # reaches the actionable CLINotFoundError below.
174
+ home = home_dir
175
+ under_home = ->(rel) { File.join(home, rel) if home }
156
176
  locations = [
157
- File.join(Dir.home, '.claude/local/claude'), # Claude Code default install location
158
- File.join(Dir.home, '.npm-global/bin/claude'),
177
+ under_home.call('.claude/local/claude'), # Claude Code default install location
178
+ under_home.call('.npm-global/bin/claude'),
159
179
  '/usr/local/bin/claude',
160
- File.join(Dir.home, '.local/bin/claude'),
161
- File.join(Dir.home, 'node_modules/.bin/claude'),
162
- File.join(Dir.home, '.yarn/bin/claude')
163
- ]
180
+ under_home.call('.local/bin/claude'),
181
+ under_home.call('node_modules/.bin/claude'),
182
+ under_home.call('.yarn/bin/claude')
183
+ ].compact
164
184
 
165
185
  locations.each do |path|
166
- return path if File.exist?(path) && File.file?(path)
186
+ # Same test as the CLAUDE_CLI_PATH branch: a non-executable file here
187
+ # would otherwise be accepted and fail at spawn with a raw EACCES
188
+ # instead of CLINotFoundError's install instructions.
189
+ return path if File.file?(path) && File.executable?(path)
167
190
  end
168
191
 
169
192
  raise CLINotFoundError.new(
@@ -174,7 +197,7 @@ module ClaudeAgentSDK
174
197
  "\n\nOr provide the path via ClaudeAgentOptions:\n" \
175
198
  " ClaudeAgentOptions.new(cli_path: '/path/to/claude')" \
176
199
  "\n\nFor hermetic deploys (Docker/CI), vendor a pinned CLI into the project:\n" \
177
- " ClaudeAgentSDK::CLIInstaller.install(version: '2.1.220')" \
200
+ " ClaudeAgentSDK::CLIInstaller.install_pinned # installs #{CLIInstaller::PINNED_CLI_VERSION}" \
178
201
  "\n\nOr point the SDK at an existing binary:\n" \
179
202
  " export #{CLI_PATH_ENV_VAR}=/path/to/claude"
180
203
  )
@@ -314,6 +337,11 @@ module ClaudeAgentSDK
314
337
  return unless @stderr
315
338
 
316
339
  @stderr.each_line("\n", @max_buffer_size + 1) do |line|
340
+ # Scrubbed at read time like stdout frames and the version probe: the
341
+ # CLI (or a tool it runs) can emit invalid UTF-8 on stderr, and an
342
+ # invalid string handed to the callback or kept for ProcessError#stderr
343
+ # raises later in the user's encoding work (JSON logging/exporters).
344
+ line = line.scrub unless line.valid_encoding?
317
345
  line_str = line.chomp
318
346
  next if line_str.empty?
319
347
 
@@ -352,6 +380,7 @@ module ClaudeAgentSDK
352
380
  return unless @stderr
353
381
 
354
382
  @stderr.each_line("\n", @max_buffer_size + 1) do |line|
383
+ line = line.scrub unless line.valid_encoding? # see #handle_stderr
355
384
  line_str = line.chomp
356
385
  next if line_str.empty?
357
386
 
@@ -375,8 +404,8 @@ module ClaudeAgentSDK
375
404
  # interpreter exit (the at_exit reaper fires only then, TERM only).
376
405
  # Nothing here may suspend: a synchronous TERM plus a plain
377
406
  # background thread for the KILL escalation. On this path the
378
- # process deliberately STAYS in the at_exit registry as a second
379
- # safety net; the normal path deregisters in teardown_process.
407
+ # process stays in the at_exit registry until the fallback confirms
408
+ # it was reaped; the normal path deregisters in teardown_process.
380
409
  force_terminate_in_background(process) unless process_teardown_complete
381
410
 
382
411
  # Snapshot-then-nil BEFORE the best-effort pipe close below: once the
@@ -407,9 +436,24 @@ module ClaudeAgentSDK
407
436
  # Cancellation can land before teardown_process reached the pipe
408
437
  # closes. stdout/stderr are read ends (close never blocks); stdin's
409
438
  # implicit flush is a no-op in practice because #write flushes
410
- # after every write. Best-effort: an IO that is already closed or
411
- # fails to close is left to GC, which the nil-ing above enables.
412
- [stdin_io, stdout_io, stderr_io].each do |io|
439
+ # after every write. A fiber writer still parked on it is woken
440
+ # first — a scheduler hand-off (like close_now's child-task stops)
441
+ # that resumes this fiber one reactor tick later; a second
442
+ # cancellation landing there only skips the closes below, which
443
+ # GC then finishes (termination already ran above). A plain
444
+ # thread parked there makes close_stdin_io use a detached thread
445
+ # (wait: false), so the close itself never blocks. Best-effort:
446
+ # an IO that is already closed or fails to close is left to GC,
447
+ # which the nil-ing above enables.
448
+ if stdin_io
449
+ begin
450
+ wake_parked_fiber_writers
451
+ close_stdin_io(stdin_io, wait: false)
452
+ rescue StandardError
453
+ nil
454
+ end
455
+ end
456
+ [stdout_io, stderr_io].each do |io|
413
457
  io&.close
414
458
  rescue StandardError
415
459
  nil
@@ -431,22 +475,18 @@ module ClaudeAgentSDK
431
475
  @stderr_task.kill
432
476
  @stderr_task.join(1)
433
477
  rescue StandardError => e
478
+ raise if e.is_a?(Async::TimeoutError)
479
+
434
480
  cleanup_errors << "stderr thread: #{e.message}"
435
481
  end
436
482
  end
437
483
 
438
- # Close stdin under the same lock that guards write — otherwise a
439
- # concurrent writer (callbacks running on FiberBoundary threads) can
440
- # see @stdin nilled mid-write and hit NoMethodError on nil.
441
- @stdin_mutex.synchronize do
442
- begin
443
- @stdin&.close
444
- rescue IOError
445
- # Already closed, ignore
446
- rescue StandardError => e
447
- cleanup_errors << "stdin: #{e.message}"
448
- end
449
- @stdin = nil
484
+ begin
485
+ shutdown_stdin
486
+ rescue StandardError => e
487
+ raise if e.is_a?(Async::TimeoutError)
488
+
489
+ cleanup_errors << "stdin: #{e.message}"
450
490
  end
451
491
 
452
492
  begin
@@ -454,6 +494,8 @@ module ClaudeAgentSDK
454
494
  rescue IOError
455
495
  # Already closed, ignore
456
496
  rescue StandardError => e
497
+ raise if e.is_a?(Async::TimeoutError)
498
+
457
499
  cleanup_errors << "stdout: #{e.message}"
458
500
  end
459
501
 
@@ -462,6 +504,8 @@ module ClaudeAgentSDK
462
504
  rescue IOError
463
505
  # Already closed, ignore
464
506
  rescue StandardError => e
507
+ raise if e.is_a?(Async::TimeoutError)
508
+
465
509
  cleanup_errors << "stderr: #{e.message}"
466
510
  end
467
511
 
@@ -482,6 +526,8 @@ module ClaudeAgentSDK
482
526
  Process.kill('KILL', @process.pid)
483
527
  @process.value
484
528
  rescue StandardError => e
529
+ raise if e.is_a?(Async::TimeoutError)
530
+
485
531
  cleanup_errors << "force kill: #{e.message}"
486
532
  end
487
533
  rescue Errno::ESRCH
@@ -490,6 +536,10 @@ module ClaudeAgentSDK
490
536
  rescue Errno::ESRCH
491
537
  # Process already dead, ignore
492
538
  rescue StandardError => e
539
+ # An outer reactor deadline is cancellation, not a cleanup warning.
540
+ # Let close's ensure retain ownership and start fallback termination.
541
+ raise if e.is_a?(Async::TimeoutError)
542
+
493
543
  cleanup_errors << "process termination: #{e.message}"
494
544
  end
495
545
 
@@ -510,39 +560,65 @@ module ClaudeAgentSDK
510
560
  # pid-reuse-safe: while the waiter thread reports alive (not yet reaped),
511
561
  # the pid cannot have been recycled.
512
562
  def force_terminate_in_background(process, grace_seconds: 2)
513
- return unless process&.alive?
563
+ return unless process
564
+
565
+ unless process.alive?
566
+ self.class.deregister_active_process(process)
567
+ return
568
+ end
514
569
 
515
570
  pid = process.pid
516
571
  begin
517
572
  Process.kill('TERM', pid)
573
+ rescue Errno::ESRCH
574
+ # The child exited before TERM; its waiter may still be reaping.
518
575
  rescue StandardError
519
- return # ESRCH: already gone; EPERM: not ours to signal
576
+ return # EPERM etc.: retain the live child in the at_exit registry.
520
577
  end
521
578
 
522
579
  Thread.new do
523
- sleep grace_seconds
524
580
  begin
525
- Process.kill('KILL', pid) if process.alive?
581
+ unless process.join(grace_seconds)
582
+ begin
583
+ Process.kill('KILL', pid) if process.alive?
584
+ rescue Errno::ESRCH
585
+ # Still wait for the waiter when exit raced the signal.
586
+ end
587
+ process.join(grace_seconds)
588
+ end
526
589
  rescue StandardError
527
- nil # died inside the grace window
590
+ nil # best-effort; retain ownership if termination/reaping failed
591
+ ensure
592
+ self.class.deregister_active_process(process) unless process.alive?
528
593
  end
529
594
  end
530
595
  end
531
596
 
532
597
  # Wait for the spawned process to exit, up to +timeout_seconds+. Polls
533
- # @process.alive? rather than using stdlib Timeout.timeout, which raises
598
+ # process.alive? rather than using stdlib Timeout.timeout, which raises
534
599
  # across threads via Thread#raise and corrupts Async fiber-scheduler state
535
600
  # (close is always called inside an Async task). Yields to the current
536
601
  # Async task when one is active so the reactor keeps running.
537
- def wait_process_with_timeout(timeout_seconds)
602
+ def wait_process_with_timeout(timeout_seconds, process = @process)
538
603
  deadline = Process.clock_gettime(Process::CLOCK_MONOTONIC) + timeout_seconds
539
604
  task = defined?(Async::Task) ? Async::Task.current? : nil
540
- while @process.alive?
605
+ while process.alive?
541
606
  raise Timeout::Error if Process.clock_gettime(Process::CLOCK_MONOTONIC) >= deadline
542
607
 
543
608
  task ? task.sleep(0.05) : sleep(0.05)
544
609
  end
545
- @process.value
610
+ process.value
611
+ end
612
+
613
+ # Polls (via #wait_process_with_timeout) instead of
614
+ # Process::Waiter#join(timeout): under a Fiber scheduler Ruby 3.2's
615
+ # Thread#join ignores its timeout and never returns for a live thread
616
+ # (probed on 3.2.0; 3.3/3.4 honor it).
617
+ def process_exited_within?(process, seconds)
618
+ wait_process_with_timeout(seconds, process)
619
+ true
620
+ rescue Timeout::Error
621
+ false
546
622
  end
547
623
 
548
624
  def write(data)
@@ -559,9 +635,17 @@ module ClaudeAgentSDK
559
635
  # underlying stream and Ruby raises IOError("stream closed in another
560
636
  # thread") inside @stdin.write — the rescue below converts that into a
561
637
  # standard CLIConnectionError so callers see a clean shutdown error.
638
+ # A fiber parked here is NOT woken by that close, though (the async
639
+ # selector never learns the fd went away), so close wakes it with the
640
+ # same IOError via the registry below — see #wake_parked_fiber_writers.
641
+ writer = Fiber.current
562
642
  stdin = @stdin_mutex.synchronize do
563
643
  raise CLIConnectionError, 'ProcessTransport is not ready for writing' unless @ready && @stdin
564
644
 
645
+ # Registered in the same critical section as the snapshot, so a
646
+ # stdin detach (which nils @stdin under this lock) sees every writer
647
+ # that still holds the IO.
648
+ @inflight_writers[writer] = Fiber.scheduler
565
649
  @stdin
566
650
  end
567
651
 
@@ -572,25 +656,37 @@ module ClaudeAgentSDK
572
656
  @ready = false
573
657
  @exit_error = CLIConnectionError.new("Failed to write to process stdin: #{e}")
574
658
  raise @exit_error
659
+ rescue Exception => e # rubocop:disable Lint/RescueException -- cancellation is not a StandardError and must keep its class
660
+ # Cancellation (Async::Stop, InlineCancellation, a private deadline
661
+ # class) delivered while parked inside IO#write on a full pipe: the
662
+ # bytes already written stay on the pipe and nothing distinguishes
663
+ # an aborted frame from a complete one, so the next well-formed
664
+ # frame would be appended to a partial one and desync the protocol
665
+ # for the rest of the session. Poison the transport and re-raise the
666
+ # ORIGINAL exception: query.rb rescues Async::Stop by class, and
667
+ # cancellation semantics depend on it propagating unchanged.
668
+ # Recovery is a new session. Plain stores, like the StandardError
669
+ # branch above — no lock, so nothing here can suspend mid-unwind.
670
+ # (A writer already queued on IO#write's internal lock still
671
+ # appends its frame after the partial one; the session is dead
672
+ # either way, and every later write fails fast on @exit_error.)
673
+ @ready = false
674
+ @exit_error ||= CLIConnectionError.new(
675
+ "stdin write interrupted by #{exception_class_name(e)}: possible partial frame"
676
+ )
677
+ raise
678
+ ensure
679
+ @inflight_writers.delete(writer)
575
680
  end
576
681
  end
577
682
 
578
683
  def end_input
579
- # Under @stdin_mutex like write/close (the transport's documented
580
- # locking protocol; Python's end_input takes _write_lock too). The
581
- # nil-guard must live INSIDE the critical section or the TOCTOU
582
- # returns. NOTE: non-reentrant — close() inlines its own stdin
583
- # handling and must never delegate here.
584
- @stdin_mutex.synchronize do
585
- return unless @stdin
586
-
587
- begin
588
- @stdin.close
589
- rescue StandardError
590
- # Ignore
591
- end
592
- @stdin = nil
593
- end
684
+ # Same path as #close's stdin step: detach under @stdin_mutex (the
685
+ # transport's documented locking protocol; Python's end_input takes
686
+ # _write_lock too), then wake and close outside it.
687
+ shutdown_stdin
688
+ rescue StandardError
689
+ # Ignore
594
690
  end
595
691
 
596
692
  def read_messages(&block)
@@ -599,6 +695,10 @@ module ClaudeAgentSDK
599
695
  raise CLIConnectionError, 'Not connected' unless @process && @stdout
600
696
 
601
697
  json_buffer = ''
698
+ # True only when stdout reached EOF on its own. When the loop is cut
699
+ # short by close() (IOError) a partial trailing frame is expected and
700
+ # is not reported.
701
+ clean_eof = false
602
702
 
603
703
  begin
604
704
  # The limit bounds per-read allocation: a line longer than
@@ -663,6 +763,7 @@ module ClaudeAgentSDK
663
763
  next
664
764
  end
665
765
  end
766
+ clean_eof = true
666
767
  rescue IOError
667
768
  # Stream closed
668
769
  rescue StopIteration
@@ -672,10 +773,34 @@ module ClaudeAgentSDK
672
773
  # Check process completion. @process may already be nil (close() ran
673
774
  # concurrently and reset it) or already waited on (Errno::ECHILD on
674
775
  # double-wait). Both are non-fatal — the message loop just exits.
776
+ # Snapshot: a concurrent close() nils @process mid-wait.
777
+ process = @process
675
778
  returncode = nil
676
779
  termsig = nil
780
+ forced_exit = false
781
+ unreaped = false
677
782
  begin
678
- status = @process&.value
783
+ # Bounded wait. The unbounded #value parked the read loop forever
784
+ # behind a CLI that closed stdout and then hung — no 'end' ever
785
+ # reached query()/receive_response. Past the grace period escalate
786
+ # like #close (TERM, then KILL) and report it as an error below: a
787
+ # child that outlives its stdout is wedged, whatever its exit code.
788
+ # The poll parks only this task (task.sleep) on a reactor.
789
+ if process && !process_exited_within?(process, EOF_EXIT_GRACE_SECONDS)
790
+ forced_exit = true
791
+ begin
792
+ Process.kill('TERM', process.pid)
793
+ unless process_exited_within?(process, EOF_TERM_GRACE_SECONDS)
794
+ Process.kill('KILL', process.pid)
795
+ # Even KILL can't reap a child stuck in uninterruptible kernel
796
+ # I/O; bound this last wait too rather than block on #value.
797
+ unreaped = !process_exited_within?(process, EOF_TERM_GRACE_SECONDS)
798
+ end
799
+ rescue Errno::ESRCH
800
+ # Exited between the check and the signal; value below is final.
801
+ end
802
+ end
803
+ status = process&.value unless unreaped
679
804
  # exitstatus is nil when the child died from a signal (OOM-kill
680
805
  # SIGKILL, SIGSEGV, ...) — that end-of-stream is a TRUNCATED response,
681
806
  # not a clean success. Python surfaces it as a negative returncode.
@@ -691,23 +816,54 @@ module ClaudeAgentSDK
691
816
  # reach (e.g. a Client abandoned without #disconnect, or direct transport
692
817
  # use). Idempotent — #close's own deregister becomes a harmless no-op, and
693
818
  # #close still sees @process (left set here) for its termination logic.
694
- self.class.deregister_active_process(@process)
819
+ # A child that could not be reaped even after KILL stays registered:
820
+ # the at-exit safety net still owns it.
821
+ self.class.deregister_active_process(@process) unless unreaped
695
822
 
696
- if termsig || (returncode && returncode != 0)
823
+ if forced_exit || termsig || (returncode && returncode != 0)
697
824
  # Wait briefly for stderr thread to finish draining
698
825
  @stderr_task&.join(1)
699
826
 
700
827
  stderr_text = @recent_stderr_mutex.synchronize { @recent_stderr.last(10).join("\n") }
701
828
  stderr_text = 'No stderr output captured' if stderr_text.empty?
702
829
 
830
+ message =
831
+ if unreaped
832
+ "Command did not exit within #{EOF_EXIT_GRACE_SECONDS}s of closing stdout, and could not be " \
833
+ 'reaped even after SIGKILL'
834
+ elsif forced_exit
835
+ "Command did not exit within #{EOF_EXIT_GRACE_SECONDS}s of closing stdout; terminated by the SDK" +
836
+ (termsig ? " with signal #{termsig}" : '')
837
+ elsif termsig
838
+ "Command terminated by signal #{termsig}"
839
+ else
840
+ "Command failed with exit code #{returncode}"
841
+ end
842
+
703
843
  @exit_error = ProcessError.new(
704
- termsig ? "Command terminated by signal #{termsig}" : "Command failed with exit code #{returncode}",
844
+ message,
705
845
  # Negative-signal exit_code mirrors Python's subprocess returncode.
706
846
  exit_code: termsig ? -termsig : returncode,
707
847
  stderr: stderr_text
708
848
  )
709
849
  raise @exit_error
710
850
  end
851
+
852
+ # A clean exit with a newline-less partial frame still buffered: the
853
+ # CLI (or whatever sits between it and us) cut a message short. The
854
+ # in-loop parse never sees it complete, so it used to be dropped
855
+ # silently — a missing ResultMessage with no error. Report it like
856
+ # the over-cap path does; a bare whitespace tail is not a frame, and
857
+ # once a concurrent #close has reset the transport (process nil) the
858
+ # stream was torn down deliberately and a cut-off tail is expected.
859
+ return unless clean_eof && process && !json_buffer.strip.empty?
860
+
861
+ raise CLIJSONDecodeError.new(
862
+ json_buffer,
863
+ StandardError.new(
864
+ "stdout ended mid-frame: #{json_buffer.bytesize} bytes buffered without a terminating newline"
865
+ )
866
+ )
711
867
  end
712
868
 
713
869
  def check_claude_version
@@ -789,6 +945,101 @@ module ClaudeAgentSDK
789
945
  end
790
946
  end
791
947
 
948
+ # Detach stdin under the same lock that guards #write's readiness
949
+ # check-and-snapshot — a concurrent writer (reactor fiber, or callback
950
+ # on a FiberBoundary thread) either sees nil (raises not-ready) or is
951
+ # already registered with its own snapshot — then wake and close
952
+ # OUTSIDE the lock: waking hands control to the writer, whose unwinding
953
+ # must not find the lock held, and the close may join a helper thread.
954
+ # Detach before waking, so no writer can start after the wake. Caller
955
+ # must NOT hold @stdin_mutex (non-reentrant). Raises what IO#close
956
+ # raises, except IOError (already closed).
957
+ def shutdown_stdin
958
+ stdin_io = @stdin_mutex.synchronize do
959
+ io = @stdin
960
+ @stdin = nil
961
+ io
962
+ end
963
+ return unless stdin_io
964
+
965
+ wake_parked_fiber_writers
966
+ close_stdin_io(stdin_io)
967
+ end
968
+
969
+ # Raise IOError into writer fibers of THIS thread's scheduler that are
970
+ # parked inside the stdin IO call (a full pipe: the CLI stopped
971
+ # reading). Closing the fd does not wake them — the async selector never
972
+ # learns the fd went away, so the writer stayed parked forever (Ruby
973
+ # 3.2/3.3/3.4, probed) and on Ruby 3.2 a close from another thread even
974
+ # delivered the IOError into the scheduler loop itself. The injected
975
+ # IOError is exactly what a plain-thread writer gets from the close, so
976
+ # both unwind through #write's StandardError branch into
977
+ # CLIConnectionError — the documented shutdown behavior. (Not
978
+ # Task#stop: that would cancel the writer's whole task, which may be
979
+ # the caller's own.) Scheduler#raise is the hand-off Task#stop and
980
+ # task timeouts use: the writer runs its unwinding now and this fiber
981
+ # resumes on the next reactor tick. Residual: a writer on another
982
+ # thread's reactor — or any fiber writer when stdin is closed off-reactor
983
+ # (e.g. Query's schedulerless fallback close) — cannot be reached from
984
+ # here: it stays parked on 3.3/3.4, and on 3.2 the close crashes its
985
+ # reactor with that IOError. The SDK's own writers share close's reactor.
986
+ def wake_parked_fiber_writers
987
+ scheduler = Fiber.scheduler
988
+ return unless scheduler
989
+
990
+ # to_a: one GVL-atomic snapshot — iterating the live Hash would let a
991
+ # concurrent insert raise in the writer. Newest first: a later writer
992
+ # is queued on IO#write's internal lock behind an earlier one, and
993
+ # waking the lock holder first makes its unlock hand the lock to the
994
+ # queued writer (scheduler.unblock) — a stale wakeup that would then
995
+ # cut that writer's NEXT suspension short. Woken first, the queued
996
+ # writer leaves the lock's wait queue in its own unwinding. Re-check
997
+ # registration before each raise: a writer that already unwound must
998
+ # not be interrupted wherever it is now.
999
+ @inflight_writers.to_a.reverse_each do |fiber, owner|
1000
+ next unless @inflight_writers.key?(fiber)
1001
+ next unless owner.equal?(scheduler) && !fiber.equal?(Fiber.current) && fiber.alive?
1002
+
1003
+ scheduler.raise(fiber, IOError, 'stdin closed while a write was in progress')
1004
+ rescue FiberError
1005
+ # Finished (or resumed) meanwhile — nothing to wake.
1006
+ end
1007
+ end
1008
+
1009
+ # Close the stdin write end without stalling the reactor. On Ruby 3.3+
1010
+ # IO#close waits for threads blocked on the fd; called from a scheduler
1011
+ # fiber that wait is a scheduler sleep the blocked thread's wakeup never
1012
+ # resumes, so a FiberBoundary worker parked in #write on a full pipe
1013
+ # hung close — and the whole reactor — forever (probed on 3.3.9 and
1014
+ # 3.4.5; 3.2 does not wait). A plain helper thread has no scheduler:
1015
+ # its close interrupts the parked writer (IOError "stream closed in
1016
+ # another thread") and returns. Used only on a reactor with a
1017
+ # plain-thread writer still registered; everything else closes inline.
1018
+ # +wait+ false detaches the helper for #close's cancellation ensure,
1019
+ # which must not block. Raises what IO#close raises, except IOError
1020
+ # (already closed); a helper's own failure is dropped (best-effort).
1021
+ def close_stdin_io(io, wait: true)
1022
+ if Fiber.scheduler && @inflight_writers.value?(nil)
1023
+ helper = Thread.new do
1024
+ io.close
1025
+ rescue StandardError
1026
+ nil
1027
+ end
1028
+ helper.join if wait
1029
+ else
1030
+ io.close
1031
+ end
1032
+ rescue IOError
1033
+ # Already closed, ignore
1034
+ end
1035
+
1036
+ # An anonymous per-invocation cancellation class (FiberBoundary's
1037
+ # cooperative timeouts, Query's private deadline) has no name; report
1038
+ # its nearest named ancestor instead of "#<Class:0x...>".
1039
+ def exception_class_name(error)
1040
+ error.class.name || error.class.ancestors.find(&:name).name
1041
+ end
1042
+
792
1043
  # Append a stderr line to the recent-stderr ring, dropping the oldest
793
1044
  # entry once the buffer exceeds RECENT_STDERR_LINES_LIMIT. Used to surface the
794
1045
  # last few lines in ProcessError when the CLI exits non-zero.
@@ -798,6 +1049,20 @@ module ClaudeAgentSDK
798
1049
  @recent_stderr.shift if @recent_stderr.size > RECENT_STDERR_LINES_LIMIT
799
1050
  end
800
1051
  end
1052
+
1053
+ # The home directory for the well-known install probes, or nil when none
1054
+ # is usable. Dir.home raises ArgumentError when HOME is unset and the uid
1055
+ # has no passwd entry (docker --user in a minimal image), and returns an
1056
+ # empty or relative HOME verbatim — probing under "" or a relative path
1057
+ # would check files that are not where the user's install lives (and a
1058
+ # relative hit would be spawned from options.cwd, i.e. a different file).
1059
+ # SessionResume.home_dir applies the same rule.
1060
+ def home_dir
1061
+ home = Dir.home
1062
+ home if File.absolute_path?(home)
1063
+ rescue ArgumentError
1064
+ nil
1065
+ end
801
1066
  end
802
1067
  end
803
1068
 
@@ -197,13 +197,15 @@ module ClaudeAgentSDK
197
197
 
198
198
  # -- Optional: list_session_summaries ----------------------------------
199
199
 
200
- def check_list_session_summaries(fresh, has_list_sessions, has_delete)
200
+ def check_list_session_summaries(fresh, has_list_sessions, has_delete) # rubocop:disable Metrics/MethodLength
201
201
  # 14. persisted fold output round-trips through fold_session_summary.
202
202
  store = fresh.call
203
203
  summ_key = { 'project_key' => 'proj', 'session_id' => 'summ-sess' }
204
- store.append(summ_key, [entry('timestamp' => '2024-01-01T00:00:00.000Z', 'customTitle' => 'first'),
204
+ store.append(summ_key, [entry('type' => 'user', 'timestamp' => '2024-01-01T00:00:00.000Z',
205
+ 'customTitle' => 'first', 'message' => { 'content' => 'first prompt' }),
205
206
  entry('timestamp' => '2024-01-01T00:00:01.000Z')])
206
- store.append(summ_key, [entry('timestamp' => '2024-01-01T00:00:02.000Z', 'customTitle' => 'second')])
207
+ store.append(summ_key, [entry('type' => 'user', 'timestamp' => '2024-01-01T00:00:02.000Z',
208
+ 'customTitle' => 'second', 'message' => { 'content' => 'later prompt' })])
207
209
  store.append({ 'project_key' => 'other', 'session_id' => 'elsewhere' },
208
210
  [entry('timestamp' => '2024-01-01T00:00:00.000Z')])
209
211
 
@@ -219,9 +221,17 @@ module ClaudeAgentSDK
219
221
  end
220
222
 
221
223
  assert(summ['data'].is_a?(Hash), 'summary data must be a Hash')
224
+ # Independent expectations: refolding the adapter's own output alone
225
+ # accepts even an empty/stale sidecar and hides valid sessions in listings.
226
+ expected_data = { 'custom_title' => 'second', 'created_at' => 1_704_067_200_000,
227
+ 'first_prompt' => 'first prompt', 'first_prompt_locked' => true,
228
+ 'is_sidechain' => false }
229
+ assert_eq(summ['data'].slice(*expected_data.keys), expected_data,
230
+ 'summary data must persist latest title and first timestamp/prompt across appends')
222
231
  refolded = SessionSummary.fold_session_summary(summ, summ_key, [entry('timestamp' => '2024-01-01T00:00:03.000Z')])
223
232
  assert_eq(refolded['session_id'], 'summ-sess', 'refold must preserve session_id')
224
233
  assert_eq(refolded['mtime'], summ['mtime'], 'fold must preserve prev mtime verbatim')
234
+ assert_eq(refolded['data'].slice(*expected_data.keys), expected_data, 'refold must preserve summary data')
225
235
 
226
236
  # Subagent appends must NOT affect the main session's summary.
227
237
  store.append(summ_key.merge('subpath' => 'subagents/agent-1'),