claude-agent-sdk 1.0.0 → 1.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (47) hide show
  1. checksums.yaml +4 -4
  2. data/.yardopts +10 -0
  3. data/CHANGELOG.md +110 -0
  4. data/README.md +43 -31
  5. data/docs/cli-installer.md +26 -4
  6. data/docs/client.md +40 -11
  7. data/docs/configuration.md +206 -1
  8. data/docs/errors.md +32 -2
  9. data/docs/hooks-and-permissions.md +30 -10
  10. data/docs/mcp-servers.md +30 -9
  11. data/docs/observability.md +61 -10
  12. data/docs/options.md +232 -0
  13. data/docs/rails.md +263 -18
  14. data/docs/sessions.md +40 -12
  15. data/docs/subagents.md +1 -1
  16. data/docs/types.md +100 -11
  17. data/lib/claude_agent_sdk/cli_installer.rb +140 -19
  18. data/lib/claude_agent_sdk/command_builder.rb +84 -27
  19. data/lib/claude_agent_sdk/fiber_boundary.rb +45 -2
  20. data/lib/claude_agent_sdk/instrumentation/otel.rb +90 -28
  21. data/lib/claude_agent_sdk/query.rb +547 -132
  22. data/lib/claude_agent_sdk/railtie.rb +27 -2
  23. data/lib/claude_agent_sdk/sdk_mcp_server.rb +78 -26
  24. data/lib/claude_agent_sdk/session_mutations.rb +112 -92
  25. data/lib/claude_agent_sdk/session_resume.rb +356 -39
  26. data/lib/claude_agent_sdk/session_store.rb +31 -2
  27. data/lib/claude_agent_sdk/sessions.rb +720 -138
  28. data/lib/claude_agent_sdk/subprocess_cli_transport.rb +252 -29
  29. data/lib/claude_agent_sdk/testing/session_store_conformance.rb +18 -7
  30. data/lib/claude_agent_sdk/transcript_mirror_batcher.rb +45 -37
  31. data/lib/claude_agent_sdk/transport.rb +28 -12
  32. data/lib/claude_agent_sdk/types/attributes.rb +9 -0
  33. data/lib/claude_agent_sdk/types/base.rb +85 -15
  34. data/lib/claude_agent_sdk/types/hooks.rb +73 -0
  35. data/lib/claude_agent_sdk/types/mcp.rb +37 -1
  36. data/lib/claude_agent_sdk/types/messages.rb +7 -1
  37. data/lib/claude_agent_sdk/types/option_values.rb +186 -4
  38. data/lib/claude_agent_sdk/types/options.rb +104 -17
  39. data/lib/claude_agent_sdk/types/permissions.rb +18 -9
  40. data/lib/claude_agent_sdk/version.rb +1 -1
  41. data/lib/claude_agent_sdk.rb +111 -53
  42. data/lib/generators/claude_agent_sdk/install/templates/claude_agent_sdk.rb.tt +6 -0
  43. data/sig/claude_agent_sdk/types/hooks.rbs +6 -3
  44. data/sig/claude_agent_sdk/types/option_values.rbs +23 -6
  45. data/sig/claude_agent_sdk/types/options.rbs +20 -7
  46. data/sig/claude_agent_sdk/types/permissions.rbs +4 -2
  47. metadata +6 -4
@@ -89,6 +89,11 @@ module ClaudeAgentSDK
89
89
  LITE_READ_BUF_SIZE = 65_536
90
90
  MAX_SANITIZED_LENGTH = 200
91
91
 
92
+ # How far into a transcript the disk listing looks for the first prompt,
93
+ # and for the first timestamp, when the head window holds none (see
94
+ # first_prompt_from_file, created_at_from_file).
95
+ FIRST_PROMPT_SCAN_LIMIT = 1_048_576
96
+
92
97
  UUID_RE = /\A[0-9a-f]{8}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{4}-[0-9a-f]{12}\z/i
93
98
 
94
99
  # Subagent ids as the CLI writes them (agent-<id>.jsonl): hex ids and
@@ -139,28 +144,131 @@ module ClaudeAgentSDK
139
144
  out.join
140
145
  end
141
146
 
142
- # Sanitize a filesystem path to a project directory name
147
+ # Sanitize a filesystem path to a project directory name.
148
+ #
149
+ # The CLI does this with JavaScript's replace(/[^a-zA-Z0-9]/g, "-"),
150
+ # without the `u` flag: the replacement runs per UTF-16 code unit, so a
151
+ # character outside the BMP (an emoji, a CJK Extension B ideograph) is a
152
+ # surrogate pair and becomes TWO hyphens. One hyphen per code point named
153
+ # a directory the CLI never created — every directory-scoped session API
154
+ # came back empty for such a path, and the store key computed here did
155
+ # not match the one the transcript mirror derives from the CLI's own
156
+ # path. (The hash below already works on code units, see simple_hash.)
143
157
  def sanitize_path(name)
144
- sanitized = name.gsub(SANITIZE_RE, '-')
158
+ sanitized = name.gsub(SANITIZE_RE) { |char| char.ord > 0xFFFF ? '--' : '-' }
145
159
  return sanitized if sanitized.length <= MAX_SANITIZED_LENGTH
146
160
 
147
161
  "#{sanitized[0, MAX_SANITIZED_LENGTH]}-#{simple_hash(name)}"
148
162
  end
149
163
 
150
164
  # Resolve a directory to its canonical form (realpath + NFC), matching the
151
- # CLI's project-directory naming. Falls back to an absolute NFC path when
152
- # realpath can't resolve it (e.g. the directory does not exist yet) — Ruby's
153
- # File.realpath raises on missing paths whereas Python's os.path.realpath is
154
- # lexical for the missing suffix, so expand_path restores that behavior.
155
- # Known divergence: for a MISSING path Python still resolves symlinks in
156
- # the existing prefix (so a deleted /tmp/proj on macOS canonicalizes to
157
- # /private/tmp/proj and its project dir is found); the expand_path
158
- # fallback resolves none, so deleted-directory lookups under symlinked
159
- # prefixes can miss.
165
+ # CLI's project-directory naming.
166
+ #
167
+ # A path that cannot be resolved as a whole (the directory was removed, or
168
+ # does not exist yet) is resolved as far as it exists, the way Python's
169
+ # os.path.realpath does where Ruby's File.realpath raises
170
+ # (resolve_missing_path). The CLI keyed the project by the real path while
171
+ # the directory existed, so a removed /tmp/proj on macOS must still
172
+ # canonicalize to /private/tmp/proj for its sessions to be found; a plain
173
+ # expand_path (the earlier fallback) resolved no symlink at all. The path
174
+ # goes to that walk absolute but with its `..` components still in it
175
+ # (absolute_path_keeping_dots).
160
176
  def canonicalize_path(dir)
161
- File.realpath(dir).unicode_normalize(:nfc)
177
+ nfc_path(File.realpath(dir))
162
178
  rescue SystemCallError
163
- File.expand_path(dir).unicode_normalize(:nfc)
179
+ nfc_path(resolve_missing_path(absolute_path_keeping_dots(dir)))
180
+ end
181
+
182
+ # +dir+ as an absolute path with its `.` and `..` components left where
183
+ # they are, for resolve_missing_path. File.expand_path — what that walk
184
+ # was given before — removes a `..` together with the name in front of
185
+ # it. When that name is a symlink, the parent meant is the one of the
186
+ # link's TARGET: a session recorded through `current/../sibling` while
187
+ # `current` pointed at checkouts/project belongs to checkouts/sibling.
188
+ # Collapsed by name, the path was the sibling of the link — another
189
+ # project key as soon as the target was gone and File.realpath raised.
190
+ #
191
+ # Otherwise as File.expand_path has it: a relative path starts at the
192
+ # working directory, a leading ~ or ~user is that home directory
193
+ # (File.realpath expands neither, so `directory: '~/project'` has always
194
+ # been resolved through here), and a Pathname is taken as well as a
195
+ # String (File.path).
196
+ #
197
+ # The result is a BINARY String, and resolve_missing_path works on bytes
198
+ # throughout. Its parts come tagged by the locale — under LANG=C the
199
+ # working directory BINARY and a link target US-ASCII, whatever their
200
+ # bytes — and as tagged Strings they did not always go together: a link
201
+ # with a non-ASCII target made the walk raise there ("invalid byte
202
+ # sequence in US-ASCII"), as did a non-ASCII relative path in a non-ASCII
203
+ # working directory. nfc_path tags the result.
204
+ def absolute_path_keeping_dots(dir)
205
+ dir = File.path(dir).b
206
+ return dir if File.absolute_path?(dir)
207
+
208
+ first, rest = dir.split(File::SEPARATOR, 2)
209
+ base = first&.start_with?('~') ? File.expand_path(first).b : File.join(Dir.pwd.b, first.to_s)
210
+ rest ? File.join(base, rest) : base
211
+ end
212
+
213
+ # +path+ as an NFC-normalized UTF-8 String. Paths are UTF-8 whatever the
214
+ # locale says, but under LANG=C Ruby hands out what it gets from the
215
+ # system — ENV values, Dir.pwd, File.realpath — tagged BINARY or US-ASCII.
216
+ # String#unicode_normalize raises on the first ("Unicode Normalization
217
+ # not appropriate for ASCII-8BIT": `directory: Dir.pwd` and a non-ASCII
218
+ # CLAUDE_CONFIG_DIR failed every disk session API there) and leaves the
219
+ # second as it is, non-ASCII bytes included. So the bytes are tagged
220
+ # UTF-8 first, and scrubbed when they are not valid UTF-8.
221
+ def nfc_path(path)
222
+ utf8 = path.encoding == Encoding::UTF_8 ? path : path.dup.force_encoding(Encoding::UTF_8)
223
+ utf8 = utf8.scrub unless utf8.valid_encoding?
224
+ utf8.unicode_normalize(:nfc)
225
+ end
226
+
227
+ # How many symlinks resolve_missing_path follows before it keeps a link as
228
+ # written: the guard against links that point at each other.
229
+ MAX_SYMLINK_HOPS = 40
230
+
231
+ # Resolve an absolute +path+ that does not exist as a whole, component by
232
+ # component: a component that is a symlink is followed (lstat/readlink)
233
+ # whether or not its target exists, any other one — existing or missing —
234
+ # is kept as written. So is everything after the first missing
235
+ # component, and a link past MAX_SYMLINK_HOPS. Following a link whose
236
+ # target is gone is the point: a session recorded through a symlinked
237
+ # project directory is keyed by the target, and must still be found
238
+ # through the link after the target was removed.
239
+ #
240
+ # A `..` drops the last component of what is resolved so far — after the
241
+ # links in front of it were followed, never before: the order of the
242
+ # kernel and of Python's os.path.realpath.
243
+ def resolve_missing_path(path)
244
+ root = path[%r{\A(?:[A-Za-z]:)?/+}] || File::SEPARATOR
245
+ resolved = root
246
+ pending = path.delete_prefix(root).split(File::SEPARATOR).reject(&:empty?)
247
+ hops = 0
248
+ until pending.empty?
249
+ name = pending.shift
250
+ next if name == '.'
251
+
252
+ candidate = name == '..' ? File.dirname(resolved) : File.join(resolved, name)
253
+ target = name == '..' || hops >= MAX_SYMLINK_HOPS ? nil : symlink_target(candidate)
254
+ if target.nil?
255
+ resolved = candidate
256
+ next
257
+ end
258
+
259
+ hops += 1
260
+ resolved = root if target.start_with?(File::SEPARATOR)
261
+ pending.unshift(*target.split(File::SEPARATOR).reject(&:empty?))
262
+ end
263
+ resolved
264
+ end
265
+
266
+ # The target of +path+ when it is a symlink (dangling or not), else nil.
267
+ # As bytes, like the path it is joined with (absolute_path_keeping_dots).
268
+ def symlink_target(path)
269
+ File.symlink?(path) ? File.readlink(path).b : nil
270
+ rescue SystemCallError
271
+ nil
164
272
  end
165
273
 
166
274
  # Derive the SessionStore +project_key+ for a directory (default: cwd).
@@ -186,7 +294,7 @@ module ClaudeAgentSDK
186
294
  # raised a bare ArgumentError from deep inside every disk session API.
187
295
  def config_dir
188
296
  dir = ENV.fetch('CLAUDE_CONFIG_DIR', nil)
189
- return dir.unicode_normalize(:nfc) if dir && !dir.empty?
297
+ return nfc_path(dir) if dir && !dir.empty?
190
298
 
191
299
  home = home_dir
192
300
  unless home
@@ -196,7 +304,7 @@ module ClaudeAgentSDK
196
304
  'Set CLAUDE_CONFIG_DIR to the directory holding your Claude Code data (normally ~/.claude).'
197
305
  end
198
306
 
199
- File.join(home, '.claude').unicode_normalize(:nfc)
307
+ nfc_path(File.join(home, '.claude'))
200
308
  end
201
309
 
202
310
  # A usable home directory, or nil when there is none. The ONE definition
@@ -236,19 +344,89 @@ module ClaudeAgentSDK
236
344
  sanitized = sanitize_path(path)
237
345
  exact_path = File.join(projects_dir, sanitized)
238
346
  return exact_path if File.directory?(exact_path)
347
+ return nil unless sanitized.length > MAX_SANITIZED_LENGTH
348
+
349
+ # A long path is stored under its first 200 characters plus a hash of
350
+ # the whole path. Older CLIs hashed with Bun.hash, so a directory with
351
+ # the same prefix and another suffix may be this path's — or that of
352
+ # ANY path sharing the prefix (a sibling in a deep per-tenant tree).
353
+ # The name cannot tell them apart; a transcript inside can: accept a
354
+ # candidate only if one records the path as its cwd (recorded_cwd), and
355
+ # only when exactly one candidate does. Taking the first prefix match
356
+ # listed, read and renamed another project's sessions for a directory
357
+ # that had none of its own. A directory whose transcripts record no cwd
358
+ # is not used: no guess from the name alone. The directory returned may
359
+ # still hold sessions of other paths sharing the prefix: callers keep
360
+ # only the path's own transcripts (own_transcript?).
361
+ prefix = sanitized[0, MAX_SANITIZED_LENGTH + 1] # includes the trailing '-'
362
+ verified = Dir.children(projects_dir).select do |child|
363
+ candidate = File.join(projects_dir, child)
364
+ child.start_with?(prefix) && File.directory?(candidate) && project_dir_records_cwd?(candidate, path)
365
+ end
366
+ verified.length == 1 ? File.join(projects_dir, verified.first) : nil
367
+ end
239
368
 
240
- # For long paths, scan for prefix match
241
- if sanitized.length > MAX_SANITIZED_LENGTH
242
- prefix = sanitized[0, MAX_SANITIZED_LENGTH + 1] # includes the trailing '-'
243
- Dir.children(projects_dir).each do |child|
244
- candidate = File.join(projects_dir, child)
245
- return candidate if File.directory?(candidate) && child.start_with?(prefix)
246
- end
369
+ # Whether a session transcript in +project_dir+ was recorded for +path+.
370
+ def project_dir_records_cwd?(project_dir, path)
371
+ Dir.children(project_dir).any? do |name|
372
+ name.end_with?('.jsonl') && valid_session_id?(name.delete_suffix('.jsonl')) &&
373
+ recorded_cwd(File.join(project_dir, name)) == path
247
374
  end
375
+ rescue SystemCallError
376
+ false
377
+ end
248
378
 
379
+ # Whether +file_path+, a transcript in +project_dir+ — the directory
380
+ # find_project_dir returned for +path+ — is one of +path+'s sessions. Every
381
+ # transcript of the directory named after the path is. In a directory the
382
+ # long-path prefix fallback found, only one whose own recorded cwd is the
383
+ # path: that directory can hold the sessions of every path sharing the
384
+ # prefix, and a transcript whose cwd cannot be verified is not counted.
385
+ def own_transcript?(project_dir, file_path, path)
386
+ !prefix_fallback_dir?(project_dir, path) || recorded_cwd(file_path) == path
387
+ end
388
+
389
+ # Whether +project_dir+, the directory find_project_dir returned for
390
+ # +path+, is one the long-path prefix fallback found rather than the one
391
+ # named after the path.
392
+ def prefix_fallback_dir?(project_dir, path)
393
+ File.basename(project_dir) != sanitize_path(path)
394
+ end
395
+
396
+ # The directory a session transcript was recorded in: the first non-blank
397
+ # top-level cwd of a COMPLETE line in its first LITE_READ_BUF_SIZE bytes,
398
+ # NFC-normalized; nil when there is none (or the file cannot be read). A
399
+ # line the window cuts establishes nothing: its top-level shape cannot be
400
+ # checked, and a raw "cwd" match on it may sit inside a tool input.
401
+ def recorded_cwd(file_path)
402
+ File.open(file_path, 'rb') do |file|
403
+ head = file.read(LITE_READ_BUF_SIZE) || ''
404
+ each_parsed_entry(head, file.eof?) do |entry|
405
+ cwd = entry['cwd']
406
+ return cwd.unicode_normalize(:nfc) if cwd.is_a?(String) && presence(cwd)
407
+ end
408
+ end
409
+ nil
410
+ rescue SystemCallError
249
411
  nil
250
412
  end
251
413
 
414
+ # Yield each Hash entry parsed from a COMPLETE line of +text+, a window
415
+ # read from the start of a transcript: every line that ends in a newline,
416
+ # and the last one too when +to_eof+ (the window reaches the end of the
417
+ # file). A line that does not parse is skipped.
418
+ def each_parsed_entry(text, to_eof)
419
+ complete = to_eof ? text.bytesize : (text.byterindex("\n") || -1) + 1
420
+ text.byteslice(0, complete).each_line do |line|
421
+ entry = begin
422
+ JSON.parse(line)
423
+ rescue JSON::ParserError
424
+ next
425
+ end
426
+ yield entry if entry.is_a?(Hash)
427
+ end
428
+ end
429
+
252
430
  # Extract a JSON string field value from raw text without full JSON parse
253
431
  def extract_json_string_field(text, key, last: false)
254
432
  search_patterns = ["\"#{key}\":\"", "\"#{key}\": \""]
@@ -354,11 +532,13 @@ module ClaudeAgentSDK
354
532
  nil
355
533
  end
356
534
 
357
- # Unescape a JSON string value
535
+ # Unescape a JSON string value. A slice that does not parse as a JSON
536
+ # string (a raw control character in it) is returned as it is — tagged
537
+ # UTF-8: the windows it is cut from are binary (read_head_tail).
358
538
  def unescape_json_string(str)
359
539
  JSON.parse("\"#{str}\"")
360
540
  rescue JSON::ParserError
361
- str
541
+ str.encoding == Encoding::UTF_8 ? str : str.dup.force_encoding(Encoding::UTF_8)
362
542
  end
363
543
 
364
544
  # Python's `x or None` for the summary/title fallback chains: Ruby's ||
@@ -437,16 +617,25 @@ module ClaudeAgentSDK
437
617
  end
438
618
 
439
619
  # Extract the first meaningful user prompt from the head of a JSONL file
440
- def extract_first_prompt_from_head(head) # rubocop:disable Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity -- first-prompt skip rules, matched by the store fold
441
- command_fallback = nil
620
+ def extract_first_prompt_from_head(head)
621
+ prompt, command_fallback = first_prompt_in(head)
622
+ prompt || command_fallback || ''
623
+ end
442
624
 
443
- head.each_line do |line|
625
+ # The first real user prompt among the lines of +text+, and the name of
626
+ # the first slash command seen on the way (what a session without a real
627
+ # prompt reports): [prompt or nil, command name or nil]. +command_fallback+
628
+ # carries a name found in an earlier part of the same transcript.
629
+ def first_prompt_in(text, command_fallback = nil) # rubocop:disable Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity -- first-prompt skip rules, matched by the store fold
630
+ text.each_line do |line|
444
631
  next unless line.include?('"type":"user"') || line.include?('"type": "user"')
445
632
  next if line.include?('"tool_result"')
446
633
  next if line.include?('"isMeta":true') || line.include?('"isMeta": true')
447
634
  next if line.include?('"isCompactSummary":true') || line.include?('"isCompactSummary": true')
448
635
 
449
- entry = JSON.parse(line, symbolize_names: false)
636
+ # +text+ may be a binary window or chunk: the line becomes UTF-8 here,
637
+ # scrubbed, so the text handling below never meets a stray byte.
638
+ entry = JSON.parse(utf8_transcript_text(line), symbolize_names: false)
450
639
  texts = user_entry_texts(entry)
451
640
  next unless texts
452
641
 
@@ -461,13 +650,65 @@ module ClaudeAgentSDK
461
650
 
462
651
  next if text.match?(SKIP_FIRST_PROMPT_PATTERN)
463
652
 
464
- return text.length > 200 ? "#{text[0, 200]}…" : text
653
+ return [text.length > 200 ? "#{text[0, 200]}…" : text, command_fallback]
465
654
  end
466
655
  rescue JSON::ParserError
467
656
  next
468
657
  end
469
658
 
470
- command_fallback || ''
659
+ [nil, command_fallback]
660
+ end
661
+
662
+ # first_prompt for the disk listing: from the head window, and when that
663
+ # holds no real prompt while the file goes on, from a bounded scan past
664
+ # it. CLI 2.1.x transcripts often carry a large attachment (a
665
+ # SessionStart hook's output) before the first prompt, and an SDK prompt
666
+ # that inlines a document is one line longer than the window; the head
667
+ # alone reported nil or a slash-command name for those, and a session
668
+ # with no other summary source was not listed at all — while the store
669
+ # fold, which sees every entry, reported the prompt.
670
+ def first_prompt_from_file(file_path, head, size)
671
+ prompt, command_fallback = first_prompt_in(head)
672
+ if prompt.nil? && size > head.bytesize
673
+ limit = [size, FIRST_PROMPT_SCAN_LIMIT].min
674
+ prompt, command_fallback = first_prompt_past_head(file_path, head, limit, command_fallback)
675
+ end
676
+ prompt || command_fallback || ''
677
+ end
678
+
679
+ # Continue the first-prompt scan past +head+ (see each_line_past_head). An
680
+ # IO failure there leaves the answer the head gave.
681
+ def first_prompt_past_head(file_path, head, limit, command_fallback)
682
+ each_line_past_head(file_path, head, limit) do |line|
683
+ prompt, command_fallback = first_prompt_in(line, command_fallback)
684
+ return [prompt, command_fallback] if prompt
685
+ end
686
+ [nil, command_fallback]
687
+ end
688
+
689
+ # Yield the lines of a transcript that follow the last complete line of
690
+ # +head+, up to byte +limit+ of the file. Read in fixed-size chunks,
691
+ # never line by line: one transcript line can be gigabytes, and the
692
+ # limit has to hold before the bytes are in memory. What is yielded last
693
+ # is the final line of a file without a closing newline — or a line cut
694
+ # by the limit, which does not parse and is skipped by its consumer like
695
+ # any other bad line. An IO failure ends the read quietly: this read is
696
+ # an extra, and must not hide a session.
697
+ def each_line_past_head(file_path, head, limit, &)
698
+ offset = (head.byterindex("\n") || -1) + 1
699
+ open_line = String.new(encoding: Encoding::BINARY)
700
+ File.open(file_path, 'rb') do |file|
701
+ file.seek(offset)
702
+ while offset < limit && (chunk = file.read([LITE_READ_BUF_SIZE, limit - offset].min))
703
+ offset += chunk.bytesize
704
+ open_line << chunk
705
+ newline = open_line.rindex("\n")
706
+ open_line.slice!(0, newline + 1).each_line(&) if newline
707
+ end
708
+ yield open_line unless open_line.empty?
709
+ end
710
+ rescue SystemCallError
711
+ nil
471
712
  end
472
713
 
473
714
  # Text blocks of a genuine user entry, or nil when the line should be
@@ -535,13 +776,24 @@ module ClaudeAgentSDK
535
776
  false
536
777
  end
537
778
 
779
+ # The first and the last LITE_READ_BUF_SIZE bytes of a transcript, as
780
+ # BINARY Strings — on purpose. The field scanners step through a window
781
+ # by offset (String#index with a position, text[pos], #length), and Ruby
782
+ # keeps no character index for a UTF-8 String that is not ASCII-only:
783
+ # there every one of those steps walks the bytes from the start, so a
784
+ # scan with many matches is quadratic in the window, and one multibyte
785
+ # character anywhere in it is enough (nearly every real transcript has
786
+ # one). On bytes each step is O(1). The patterns are ASCII, JSON.parse
787
+ # reads a binary source as UTF-8, and the two places that return a raw
788
+ # slice of the window tag it UTF-8 (unescape_json_string), so every value
789
+ # that leaves the scanners is a UTF-8 String as before.
538
790
  def read_head_tail(file_path, size)
539
791
  head = tail = nil
540
792
  File.open(file_path, 'rb') do |f|
541
- head = (f.read(LITE_READ_BUF_SIZE) || '').force_encoding('UTF-8')
793
+ head = f.read(LITE_READ_BUF_SIZE) || String.new(encoding: Encoding::BINARY)
542
794
  tail = if size > LITE_READ_BUF_SIZE
543
795
  f.seek([0, size - LITE_READ_BUF_SIZE].max)
544
- (f.read(LITE_READ_BUF_SIZE) || '').force_encoding('UTF-8')
796
+ f.read(LITE_READ_BUF_SIZE) || String.new(encoding: Encoding::BINARY)
545
797
  else
546
798
  head
547
799
  end
@@ -549,22 +801,70 @@ module ClaudeAgentSDK
549
801
  [head, tail]
550
802
  end
551
803
 
552
- def build_session_info(file_path, head, tail, stat, project_path) # rubocop:disable Metrics/AbcSize, Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity -- one optional field per SDKSessionInfo attribute
553
- # User-set title (customTitle) wins over AI-generated title (aiTitle).
554
- # Consult the head only when the tail has no occurrence of that field.
555
- # Normalize blanks AFTER choosing the latest occurrence: an explicit
556
- # clearing entry must not resurrect an older title from the head.
557
- # Summary-chain fields use the top-level-verified scan: a raw byte scan
558
- # also matches these keys nested inside tool_use inputs, reporting tool
559
- # arguments as the session title/summary (and diverging from the store
560
- # fold, which reads top-level keys only).
561
- custom_title = presence(extract_top_level_string_field(tail, 'customTitle', last: true) ||
562
- extract_top_level_string_field(head, 'customTitle', last: true)) ||
563
- presence(extract_top_level_string_field(tail, 'aiTitle', last: true) ||
564
- extract_top_level_string_field(head, 'aiTitle', last: true))
804
+ # [title, first prompt] of a transcript on disk, each nil when absent: the
805
+ # two values a disk listing reports as custom_title and first_prompt, and
806
+ # the ones fork_session names a fork after (SessionMutations).
807
+ #
808
+ # Title: the user-set title (customTitle) wins over the AI-generated one
809
+ # (aiTitle). The head is consulted only when the tail has no occurrence
810
+ # of the field, and blanks are normalized AFTER the latest occurrence was
811
+ # chosen (display_title): an explicit clearing entry must not resurrect an
812
+ # older title from the head. The top-level-verified scan is used: a raw
813
+ # byte scan also matches these keys nested inside tool_use inputs,
814
+ # reporting tool arguments as the session title (and diverging from the
815
+ # store fold, which reads top-level keys only).
816
+ def title_and_first_prompt(file_path, head, tail, size)
817
+ custom, generated = %w[customTitle aiTitle].map do |key|
818
+ extract_top_level_string_field(tail, key, last: true) || extract_top_level_string_field(head, key, last: true)
819
+ end
565
820
  # nil, not '', when there is no prompt — the store path's answer, and
566
821
  # Python's (`_extract_first_prompt_from_head(head) or None`).
567
- first_prompt = presence(extract_first_prompt_from_head(head))
822
+ [display_title(custom, generated), presence(first_prompt_from_file(file_path, head, size))]
823
+ end
824
+
825
+ # The ONE rule for a session's title, given the latest custom title and the
826
+ # latest AI title of its transcript: blank counts as absent, custom first.
827
+ # Shared by the disk listing, and by fork_session on the disk and the
828
+ # store path (the store listing applies it in SessionSummary).
829
+ def display_title(custom_title, ai_title)
830
+ presence(custom_title) || presence(ai_title)
831
+ end
832
+
833
+ # created_at (epoch ms) for the disk listing: the first top-level
834
+ # timestamp that parses — what the store fold takes. More reliable than
835
+ # stat().birthtime, which is unsupported on some filesystems. Every line
836
+ # is looked at, not only the first: the first record may be a
837
+ # metadata-only entry (e.g. permission-mode) with no timestamp field, and
838
+ # the first user/assistant record that follows carries one (Python #907).
839
+ #
840
+ # Parsed top-level fields of COMPLETE lines only, never a raw match: a
841
+ # file-history-snapshot entry, common near the start of an interactive
842
+ # session, has no timestamp of its own but nests one
843
+ # (snapshot.timestamp), and on a line the head window cuts a raw match
844
+ # cannot be told from such a nested one. When the complete head lines
845
+ # hold no timestamp and the file goes on, the lines past them are read
846
+ # to their end instead (bounded like the first-prompt scan): the cut line
847
+ # is usually the one that carries it — an SDK prompt of more than 64 KiB
848
+ # makes the very first line that long.
849
+ def created_at_from_file(file_path, head, size)
850
+ to_eof = size <= head.bytesize
851
+ each_parsed_entry(head, to_eof) do |entry|
852
+ created_at = parse_iso_timestamp_ms(entry['timestamp'])
853
+ return created_at if created_at
854
+ end
855
+ return nil if to_eof
856
+
857
+ each_line_past_head(file_path, head, [size, FIRST_PROMPT_SCAN_LIMIT].min) do |line|
858
+ each_parsed_entry(line, true) do |entry|
859
+ created_at = parse_iso_timestamp_ms(entry['timestamp'])
860
+ return created_at if created_at
861
+ end
862
+ end
863
+ nil
864
+ end
865
+
866
+ def build_session_info(file_path, head, tail, stat, project_path) # rubocop:disable Metrics/AbcSize -- one optional field per SDKSessionInfo attribute
867
+ custom_title, first_prompt = title_and_first_prompt(file_path, head, tail, stat.size)
568
868
  # lastPrompt tail entry shows what the user was most recently doing.
569
869
  summary = custom_title ||
570
870
  presence(extract_top_level_string_field(tail, 'lastPrompt', last: true)) ||
@@ -577,14 +877,7 @@ module ClaudeAgentSDK
577
877
  tag_line = tail.lines.reverse.find { |ln| ln.start_with?('{"type":"tag"') }
578
878
  tag_value = presence(tag_line ? extract_json_string_field(tag_line, 'tag', last: true) : nil)
579
879
 
580
- # created_at from the first ISO timestamp found in the head (epoch ms).
581
- # More reliable than stat().birthtime which is unsupported on some
582
- # filesystems. Scans the whole head rather than only the first line
583
- # because the first record may be a metadata-only entry (e.g.
584
- # permission-mode) with no timestamp field; the first user/assistant
585
- # record that follows does carry one (Python #907).
586
- first_timestamp = extract_json_string_field(head, 'timestamp', last: false)
587
- created_at = parse_iso_timestamp_ms(first_timestamp) if first_timestamp
880
+ created_at = created_at_from_file(file_path, head, stat.size)
588
881
 
589
882
  SDKSessionInfo.new(
590
883
  session_id: File.basename(file_path, '.jsonl'),
@@ -616,7 +909,10 @@ module ClaudeAgentSDK
616
909
  return nil unless timestamp_str.is_a?(String)
617
910
 
618
911
  require 'time'
619
- (Time.iso8601(timestamp_str).to_f * 1000).to_i
912
+ # Integer arithmetic: through a Float (to_f * 1000, truncated) about one
913
+ # millisecond value in eight came out 1 ms low — ...30.933Z as ...932.
914
+ time = Time.iso8601(timestamp_str)
915
+ (time.to_i * 1000) + (time.nsec / 1_000_000)
620
916
  rescue ArgumentError
621
917
  nil
622
918
  end
@@ -626,10 +922,19 @@ module ClaudeAgentSDK
626
922
  return [] unless File.directory?(project_dir)
627
923
 
628
924
  sessions = []
629
- Dir.glob(File.join(project_dir, '*.jsonl')).each do |file_path|
630
- stem = File.basename(file_path, '.jsonl')
925
+ # Listing a directory found by the long-path prefix fallback: only the
926
+ # transcripts recorded for +project_path+ (own_transcript?).
927
+ verify = project_path && prefix_fallback_dir?(project_dir, project_path)
928
+ # base:, not a pattern built from the directory: a config dir path with
929
+ # glob characters in it (`/Volumes/Data [SSD]/…`, `/srv/{tenant}/…`)
930
+ # is a path, and as part of the pattern it matched nothing.
931
+ Dir.glob('*.jsonl', base: project_dir).each do |name|
932
+ stem = File.basename(name, '.jsonl')
631
933
  next unless stem.match?(UUID_RE)
632
934
 
935
+ file_path = File.join(project_dir, name)
936
+ next if verify && recorded_cwd(file_path) != project_path
937
+
633
938
  session = read_session_lite(file_path, project_path)
634
939
  sessions << session if session
635
940
  end
@@ -721,10 +1026,12 @@ module ClaudeAgentSDK
721
1026
  # List subagent IDs recorded for a session on local disk (counterpart to
722
1027
  # list_subagents_from_store). Scans
723
1028
  # <projectDir>/<sessionId>/subagents/**/agent-<id>.jsonl, including nested
724
- # workflows/<runId>/ paths, in sorted walk order. Mirrors the Python SDK's
725
- # list_subagents (#825) — no dedupe (the store variant dedupes because
726
- # adapter subkey ordering is adapter-defined; the sorted disk walk is
727
- # already deterministic).
1029
+ # workflows/<runId>/ paths, in sorted walk order (the Python SDK's
1030
+ # list_subagents, #825). Each id once, at its first position in the walk
1031
+ # — the transcript the message and metadata readers take for it. Python
1032
+ # does not dedupe here; an id whose transcript exists both directly and
1033
+ # under workflows/<runId>/ came back twice, where the store variant
1034
+ # returns it once.
728
1035
  # @param session_id [String] The session UUID
729
1036
  # @param directory [String, nil] Working directory to search in (strictly
730
1037
  # scopes to that project + its worktrees; nil searches all projects)
@@ -735,7 +1042,7 @@ module ClaudeAgentSDK
735
1042
  subagents_dir = resolve_subagents_dir(session_id, directory)
736
1043
  return [] if subagents_dir.nil?
737
1044
 
738
- collect_agent_files(subagents_dir).map(&:first)
1045
+ collect_agent_files(subagents_dir).map(&:first).uniq
739
1046
  end
740
1047
 
741
1048
  # Read the optional subagent metadata sidecar without reading its transcript.
@@ -831,8 +1138,13 @@ module ClaudeAgentSDK
831
1138
  # List sessions from a SessionStore. Store-backed counterpart to
832
1139
  # list_sessions. Uses the store's incremental summaries (one batch call +
833
1140
  # gap-fill) when available, else falls back to list_sessions + one load per
834
- # session. Sessions are derived through the same fold the disk path uses, so
835
- # both paths agree for identical transcript content.
1141
+ # session. Sessions are derived by folding EVERY entry
1142
+ # (SessionSummary.fold_session_summary), field by field under the rules
1143
+ # of the disk reader — which only reads the head and tail windows of a
1144
+ # transcript (plus the bounded first-prompt scan). The two paths agree
1145
+ # for identical transcript content unless the entry deciding a field
1146
+ # lies outside those windows; docs/sessions.md ("Listing Sessions") lists
1147
+ # the cases.
836
1148
  #
837
1149
  # @param session_store [SessionStore] store implementing list_session_summaries and/or list_sessions
838
1150
  # @return [Array<SDKSessionInfo>] sorted by last_modified descending
@@ -841,17 +1153,18 @@ module ClaudeAgentSDK
841
1153
  project_path = canonicalize_path(directory.nil? ? '.' : directory.to_s)
842
1154
  project_key = sanitize_path(project_path)
843
1155
 
844
- if SessionStore.implements?(session_store, :list_session_summaries)
845
- via = list_sessions_via_summaries(session_store, project_key, project_path, limit, offset)
846
- return via unless via.nil?
847
- end
1156
+ via = list_sessions_via_summaries(session_store, project_key, project_path, limit, offset)
1157
+ return via unless via.nil?
848
1158
 
849
- unless SessionStore.implements?(session_store, :list_sessions)
1159
+ listed, listing = SessionStores.optional_call(session_store, :list_sessions) do
1160
+ session_store.list_sessions(project_key)
1161
+ end
1162
+ unless listed
850
1163
  raise ArgumentError,
851
1164
  'session_store implements neither list_session_summaries nor list_sessions -- cannot list sessions'
852
1165
  end
853
1166
 
854
- listing = Array(session_store.list_sessions(project_key))
1167
+ listing = Array(listing)
855
1168
  # Build all-placeholder slots (the shape the summaries fast path uses) and
856
1169
  # reuse its bounded pagination: sessions are loaded newest-first only
857
1170
  # until the page fills (~offset + limit + dropped), instead of one full
@@ -874,10 +1187,12 @@ module ClaudeAgentSDK
874
1187
  return nil unless valid_session_id?(session_id)
875
1188
 
876
1189
  project_path = canonicalize_path(directory.nil? ? '.' : directory.to_s)
877
- entries = session_store.load('project_key' => sanitize_path(project_path), 'session_id' => session_id)
1190
+ project_key = sanitize_path(project_path)
1191
+ entries = session_store.load('project_key' => project_key, 'session_id' => session_id)
878
1192
  return nil if entries.nil? || entries.empty?
879
1193
 
880
- derive_info_from_entries(session_id, entries, mtime_from_entries(entries), project_path)
1194
+ mtime = store_session_mtime(session_store, project_key, session_id) || mtime_from_entries(entries)
1195
+ derive_info_from_entries(session_id, entries, mtime, project_path)
881
1196
  end
882
1197
 
883
1198
  # Read a session's conversation messages from a SessionStore. Store-backed
@@ -897,16 +1212,21 @@ module ClaudeAgentSDK
897
1212
  def list_subagents_from_store(session_store:, session_id:, directory: nil)
898
1213
  return [] unless valid_session_id?(session_id)
899
1214
 
900
- unless SessionStore.implements?(session_store, :list_subkeys)
1215
+ project_key = project_key_for_directory(directory)
1216
+ implemented, subkeys = SessionStores.optional_call(session_store, :list_subkeys) do
1217
+ session_store.list_subkeys('project_key' => project_key, 'session_id' => session_id)
1218
+ end
1219
+ unless implemented
901
1220
  raise ArgumentError,
902
1221
  'session_store does not implement list_subkeys -- cannot list subagents'
903
1222
  end
904
1223
 
905
- project_key = project_key_for_directory(directory)
906
- subkeys = Array(session_store.list_subkeys('project_key' => project_key, 'session_id' => session_id))
907
1224
  seen = {}
908
- subkeys.filter_map do |subpath|
909
- next unless subpath.start_with?('subagents/')
1225
+ Array(subkeys).filter_map do |subpath|
1226
+ # A non-String subkey (Symbol, nil, Integer) is an adapter contract
1227
+ # violation; skip it like resume does instead of calling String
1228
+ # methods on it.
1229
+ next unless subpath.is_a?(String) && subpath.start_with?('subagents/')
910
1230
 
911
1231
  last = subpath.rpartition('/').last
912
1232
  next unless last.start_with?('agent-')
@@ -996,22 +1316,25 @@ module ClaudeAgentSDK
996
1316
  # -- Private helpers --
997
1317
 
998
1318
  # Summary fast-path for list_sessions_from_store. Returns the paginated
999
- # result, or nil if the store's list_session_summaries raises
1000
- # NotImplementedError (caller falls back to the slow path). Sessions missing
1319
+ # result, or nil if the store does not implement list_session_summaries
1320
+ # (see SessionStores.optional_call; the caller falls back to the slow
1321
+ # path). Sessions missing
1001
1322
  # a sidecar or whose sidecar is stale (summary.mtime < the session's current
1002
1323
  # mtime) are routed through gap-fill so the fold is recomputed from source.
1003
- def list_sessions_via_summaries(store, project_key, project_path, limit, offset) # rubocop:disable Metrics/AbcSize, Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity -- fast path plus stale/missing-sidecar gap-fill
1004
- begin
1005
- # Array(): a non-conformant store returning nil (e.g. a NULL JSONB read)
1006
- # degrades to gap-fill instead of crashing on nil.each, matching the
1007
- # defensive Array() already applied to list_sessions / list_subkeys.
1008
- summaries = Array(store.list_session_summaries(project_key))
1009
- rescue NotImplementedError
1010
- return nil
1324
+ def list_sessions_via_summaries(store, project_key, project_path, limit, offset) # rubocop:disable Metrics/AbcSize -- fast path plus stale/missing-sidecar gap-fill
1325
+ implemented, summaries = SessionStores.optional_call(store, :list_session_summaries) do
1326
+ store.list_session_summaries(project_key)
1011
1327
  end
1012
-
1013
- has_list_sessions = SessionStore.implements?(store, :list_sessions)
1014
- listing = has_list_sessions ? Array(store.list_sessions(project_key)) : []
1328
+ return nil unless implemented
1329
+
1330
+ # Array(): a non-conformant store returning nil (e.g. a NULL JSONB read)
1331
+ # degrades to gap-fill instead of crashing on nil.each, matching the
1332
+ # defensive Array() already applied to list_sessions / list_subkeys.
1333
+ summaries = Array(summaries)
1334
+ has_list_sessions, listing = SessionStores.optional_call(store, :list_sessions) do
1335
+ store.list_sessions(project_key)
1336
+ end
1337
+ listing = Array(listing)
1015
1338
  known_mtimes = listing.to_h { |e| [e['session_id'], e['mtime']] }
1016
1339
 
1017
1340
  slots = []
@@ -1098,6 +1421,27 @@ module ClaudeAgentSDK
1098
1421
  SessionSummary.summary_entry_to_sdk_info(summary, project_path)
1099
1422
  end
1100
1423
 
1424
+ # The adapter's own mtime for one session, as its listing reports it, or
1425
+ # nil when the store cannot be asked (it implements neither listing
1426
+ # method) or does not list the session.
1427
+ #
1428
+ # get_session_info(session_store:) stamps this as last_modified: it is the
1429
+ # clock list_sessions(session_store:) reports and orders by, and what the
1430
+ # docs promise on the store paths. The entries' own timestamps are another
1431
+ # clock — and absent from metadata entries, which gave last_modified 0 for
1432
+ # a session the listing showed with a real mtime. They remain the fallback
1433
+ # (mtime_from_entries) for a store with nothing but #append and #load.
1434
+ def store_session_mtime(store, project_key, session_id)
1435
+ listed, rows = SessionStores.optional_call(store, :list_sessions) { store.list_sessions(project_key) }
1436
+ unless listed
1437
+ _, rows = SessionStores.optional_call(store, :list_session_summaries) do
1438
+ store.list_session_summaries(project_key)
1439
+ end
1440
+ end
1441
+ row = Array(rows).find { |candidate| candidate.is_a?(Hash) && candidate['session_id'] == session_id }
1442
+ row && row['mtime']
1443
+ end
1444
+
1101
1445
  # Last parseable entry timestamp (epoch ms), scanning from the tail; 0 if none.
1102
1446
  def mtime_from_entries(entries)
1103
1447
  entries.reverse_each do |entry|
@@ -1159,25 +1503,33 @@ module ClaudeAgentSDK
1159
1503
  end
1160
1504
 
1161
1505
  # Find the last user/assistant entry and walk parentUuid links back to the
1162
- # root (subagent transcripts are linear). Mirrors Python's
1163
- # _build_subagent_chain.
1506
+ # root, as Python's _build_subagent_chain does. Subagent transcripts are
1507
+ # not linear either (Python's comment says they are): parallel tool calls
1508
+ # fan out exactly as in a main transcript, so the results the walk passes
1509
+ # by are put back. No flag rejection there — every subagent entry is a
1510
+ # sidechain entry.
1164
1511
  def build_subagent_chain(entries)
1165
1512
  return [] if entries.empty?
1166
1513
 
1167
1514
  by_uuid = entries.to_h { |e| [e['uuid'], e] }
1168
1515
  leaf = entries.reverse_each.find { |e| %w[user assistant].include?(e['type']) }
1169
- leaf ? walk_to_root(by_uuid, leaf) : []
1516
+ return [] unless leaf
1517
+
1518
+ reattach_parallel_tool_results(walk_to_root(by_uuid, leaf), entries, skip_flagged: false)
1170
1519
  end
1171
1520
 
1172
1521
  # Find the subpath for a subagent, scanning subkeys (subagents may be nested
1173
1522
  # under subagents/workflows/<runId>/agent-<id>) when list_subkeys is
1174
1523
  # available, else falling back to the direct subagents/agent-<id> path.
1175
1524
  def resolve_subagent_subpath(store, project_key, session_id, agent_id)
1176
- return "subagents/agent-#{agent_id}" unless SessionStore.implements?(store, :list_subkeys)
1525
+ implemented, subkeys = SessionStores.optional_call(store, :list_subkeys) do
1526
+ store.list_subkeys('project_key' => project_key, 'session_id' => session_id)
1527
+ end
1528
+ return "subagents/agent-#{agent_id}" unless implemented
1177
1529
 
1178
1530
  target = "agent-#{agent_id}"
1179
- matches = Array(store.list_subkeys('project_key' => project_key, 'session_id' => session_id))
1180
- .select { |sk| sk.start_with?('subagents/') && sk.rpartition('/').last == target }
1531
+ matches = Array(subkeys)
1532
+ .select { |sk| sk.is_a?(String) && sk.start_with?('subagents/') && sk.rpartition('/').last == target }
1181
1533
  # Several subpaths can share a trailing agent-<id> (a top-level agent and a
1182
1534
  # nested subagents/workflows/<run>/agent-<id>). Prefer the canonical
1183
1535
  # top-level path, else pick deterministically (shortest, then lexical) so
@@ -1252,11 +1604,15 @@ module ClaudeAgentSDK
1252
1604
  def append_jsonl_file_in_batches(file_path, key, store, batch_size)
1253
1605
  batch = []
1254
1606
  nbytes = 0
1255
- # encoding: transcripts are UTF-8 regardless of locale; without it a
1256
- # LANG=C process raises Encoding::InvalidByteSequenceError on the first
1257
- # multibyte line, aborting the import mid-way (Python pins utf-8 here).
1258
- File.foreach(file_path, encoding: 'UTF-8').with_index(1) do |line, lineno|
1259
- line = line.chomp
1607
+ # Read as UTF-8 bytes regardless of locale (a LANG=C process raised
1608
+ # Encoding::InvalidByteSequenceError on the first multibyte line,
1609
+ # aborting the import mid-way; Python pins utf-8 here), and scrub a
1610
+ # line that is not valid UTF-8: Ruby's JSON parser accepts a raw
1611
+ # invalid byte inside a string, and the entry it yields makes every
1612
+ # adapter that serializes what it is given raise JSON::GeneratorError
1613
+ # from #append — after the batches before it were already stored.
1614
+ File.foreach(file_path, mode: 'rb').with_index(1) do |line, lineno|
1615
+ line = utf8_transcript_text(line).chomp
1260
1616
  next if line.empty?
1261
1617
 
1262
1618
  begin
@@ -1314,7 +1670,7 @@ module ClaudeAgentSDK
1314
1670
  # os.path.realpath never raises here.
1315
1671
  canonical = canonicalize_path(directory)
1316
1672
  project_dir = find_project_dir(canonical)
1317
- if project_dir
1673
+ if project_dir && own_transcript?(project_dir, File.join(project_dir, file_name), canonical)
1318
1674
  info = read_session_lite(File.join(project_dir, file_name), canonical)
1319
1675
  return info if info
1320
1676
  end
@@ -1325,7 +1681,7 @@ module ClaudeAgentSDK
1325
1681
  next if wt_path == canonical
1326
1682
 
1327
1683
  wt_project_dir = find_project_dir(wt_path)
1328
- next unless wt_project_dir
1684
+ next unless wt_project_dir && own_transcript?(wt_project_dir, File.join(wt_project_dir, file_name), wt_path)
1329
1685
 
1330
1686
  info = read_session_lite(File.join(wt_project_dir, file_name), wt_path)
1331
1687
  return info if info
@@ -1344,13 +1700,29 @@ module ClaudeAgentSDK
1344
1700
  return project_dir ? read_sessions_from_dir(project_dir, path) : []
1345
1701
  end
1346
1702
 
1347
- # Multiple worktrees: scan all project dirs for matches
1703
+ # Several worktrees: the caller's own directory first, unconditionally.
1704
+ # `git worktree list` reports worktree ROOTS, so a subdirectory (a
1705
+ # monorepo package) is none of them, and reading only the listed paths
1706
+ # left out exactly the sessions that were asked for (Python:
1707
+ # "Always include the user's actual directory"). Then every worktree.
1708
+ #
1709
+ # A project dir named after its path is read once. One found by the
1710
+ # long-path prefix fallback is read once per path: reading it keeps
1711
+ # that path's own transcripts only (read_sessions_from_dir), and
1712
+ # worktrees whose paths share the first 200 characters share the
1713
+ # directory. Marked as read after the first of them, it never gave the
1714
+ # sessions of the others.
1348
1715
  all_sessions = []
1349
- worktree_paths.each do |wt_path|
1350
- project_dir = find_project_dir(wt_path)
1351
- next unless project_dir
1716
+ seen = {}
1717
+ [path, *worktree_paths].each do |dir|
1718
+ project_dir = find_project_dir(dir)
1719
+ next if project_dir.nil?
1720
+
1721
+ scan = prefix_fallback_dir?(project_dir, dir) ? [project_dir, dir] : project_dir
1722
+ next if seen[scan]
1352
1723
 
1353
- all_sessions.concat(read_sessions_from_dir(project_dir, wt_path))
1724
+ seen[scan] = true
1725
+ all_sessions.concat(read_sessions_from_dir(project_dir, dir))
1354
1726
  end
1355
1727
 
1356
1728
  deduplicate_sessions(all_sessions)
@@ -1376,10 +1748,11 @@ module ClaudeAgentSDK
1376
1748
  # One entry per session_id when the same session sits in several project
1377
1749
  # dirs (copied config dirs, worktrees). The newest last_modified wins; on
1378
1750
  # equal mtimes the larger file (the more complete copy), and then the
1379
- # copy scanned first — project dirs in name order for the global listing,
1380
- # worktrees in `git worktree list` order (main worktree first) for a
1381
- # directory listing. Python keeps the first copy seen in iterdir() order
1382
- # (sessions.py _deduplicate_by_session_id), which is arbitrary on a tie.
1751
+ # copy scanned first — project dirs in name order for the global listing;
1752
+ # for a directory listing the directory itself, then its worktrees in
1753
+ # `git worktree list` order (main worktree first). Python keeps the first
1754
+ # copy seen in iterdir() order (sessions.py _deduplicate_by_session_id),
1755
+ # which is arbitrary on a tie.
1383
1756
  def deduplicate_sessions(sessions)
1384
1757
  by_id = {}
1385
1758
  sessions.each do |s|
@@ -1404,6 +1777,7 @@ module ClaudeAgentSDK
1404
1777
  def detect_worktrees(path) # rubocop:disable Metrics/AbcSize, Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity -- bounded git subprocess: drained pipes, deadline kill
1405
1778
  stdin, stdout, stderr, wait_thr = Open3.popen3('git', '-C', path, 'worktree', 'list', '--porcelain')
1406
1779
  stdin.close
1780
+ stdout.binmode # bytes: no transcoding to Encoding.default_internal; tagged in worktree_paths
1407
1781
 
1408
1782
  # Drain stdout/stderr concurrently — without this, a repo with enough
1409
1783
  # worktrees to overrun the 64 KB pipe buffer causes git to block on
@@ -1433,9 +1807,7 @@ module ClaudeAgentSDK
1433
1807
 
1434
1808
  return [path] unless wait_thr.value.success?
1435
1809
 
1436
- paths = stdout_buf.lines.filter_map do |line|
1437
- line.strip.delete_prefix('worktree ') if line.start_with?('worktree ')
1438
- end
1810
+ paths = worktree_paths(stdout_buf)
1439
1811
  paths.empty? ? [path] : paths
1440
1812
  rescue StandardError
1441
1813
  [path]
@@ -1445,6 +1817,22 @@ module ClaudeAgentSDK
1445
1817
  [stdout, stderr].each { |io| io&.close rescue nil } # rubocop:disable Style/RescueModifier
1446
1818
  end
1447
1819
 
1820
+ # The paths in the output of `git worktree list --porcelain`.
1821
+ #
1822
+ # The output was read as bytes and is UTF-8 here, whatever the locale:
1823
+ # under LANG=C the pipe yielded US-ASCII Strings, the first non-ASCII
1824
+ # worktree path made String#strip raise, and detect_worktrees' rescue
1825
+ # then dropped EVERY worktree. Each path is NFC-normalized, like every
1826
+ # path the SDK derives a project dir name from (canonicalize_path; Python
1827
+ # normalizes here too): git prints a path as the filesystem stores it, and
1828
+ # a decomposed name sanitizes to a different project dir than the one the
1829
+ # CLI created.
1830
+ def worktree_paths(porcelain)
1831
+ porcelain.force_encoding(Encoding::UTF_8).lines.filter_map do |line|
1832
+ line.strip.delete_prefix('worktree ').unicode_normalize(:nfc) if line.start_with?('worktree ')
1833
+ end
1834
+ end
1835
+
1448
1836
  def find_session_file(session_id, directory)
1449
1837
  projects_dir = File.join(config_dir, 'projects')
1450
1838
  return nil unless File.directory?(projects_dir)
@@ -1453,13 +1841,13 @@ module ClaudeAgentSDK
1453
1841
 
1454
1842
  if directory
1455
1843
  path = canonicalize_path(directory)
1456
- found = stat_candidate(find_project_dir(path), file_name)
1844
+ found = stat_candidate(find_project_dir(path), file_name, path)
1457
1845
  return found if found
1458
1846
 
1459
1847
  detect_worktrees(path).each do |wt_path|
1460
1848
  next if wt_path == path # already tried above
1461
1849
 
1462
- found = stat_candidate(find_project_dir(wt_path), file_name)
1850
+ found = stat_candidate(find_project_dir(wt_path), file_name, wt_path)
1463
1851
  return found if found
1464
1852
  end
1465
1853
 
@@ -1485,11 +1873,16 @@ module ClaudeAgentSDK
1485
1873
  # exists AND is non-empty — a 0-byte stub in one project dir must not
1486
1874
  # stop the search when the real transcript lives under another
1487
1875
  # worktree's project dir (same hazard SessionMutations.try_append guards).
1488
- def stat_candidate(project_dir, file_name)
1876
+ # With +path+ (a directory-scoped lookup), the candidate must also be one
1877
+ # of that path's own transcripts (own_transcript?).
1878
+ def stat_candidate(project_dir, file_name, path = nil)
1489
1879
  return nil if project_dir.nil?
1490
1880
 
1491
1881
  candidate = File.join(project_dir, file_name)
1492
- File.size(candidate).positive? ? candidate : nil
1882
+ return nil unless File.size(candidate).positive?
1883
+ return nil if path && !own_transcript?(project_dir, candidate, path)
1884
+
1885
+ candidate
1493
1886
  rescue SystemCallError
1494
1887
  nil
1495
1888
  end
@@ -1529,8 +1922,8 @@ module ClaudeAgentSDK
1529
1922
  def parse_jsonl_entries(file_path)
1530
1923
  entries = []
1531
1924
 
1532
- File.foreach(file_path) do |line|
1533
- entry = JSON.parse(line.strip, symbolize_names: false)
1925
+ File.foreach(file_path, mode: 'rb') do |line|
1926
+ entry = JSON.parse(utf8_transcript_text(line).strip, symbolize_names: false)
1534
1927
  next unless entry.is_a?(Hash)
1535
1928
  next unless TRANSCRIPT_ENTRY_TYPES.include?(entry['type'])
1536
1929
  next unless entry['uuid'].is_a?(String)
@@ -1542,6 +1935,20 @@ module ClaudeAgentSDK
1542
1935
  entries
1543
1936
  end
1544
1937
 
1938
+ # Transcript text (one line, or a run of lines) read in binary mode, as
1939
+ # UTF-8. Transcripts are UTF-8 whatever the process locale says: a line
1940
+ # tagged with the locale's encoding (File.foreach's default) raised from
1941
+ # String#strip on the first non-ASCII character under LANG=C. Bytes that
1942
+ # are not valid UTF-8 — a final line the CLI was killed in the middle of,
1943
+ # raw binary in a tool result — become U+FFFD, the policy
1944
+ # SessionMutations.parse_fork_transcript already has: a torn line then
1945
+ # fails JSON.parse and is skipped like any other bad line instead of
1946
+ # raising, and a complete line keeps its entry.
1947
+ def utf8_transcript_text(text)
1948
+ text.force_encoding(Encoding::UTF_8)
1949
+ text.valid_encoding? ? text : text.scrub
1950
+ end
1951
+
1545
1952
  # Build the conversation chain by finding the leaf and walking parentUuid.
1546
1953
  # Returns messages in chronological order (root -> leaf).
1547
1954
  #
@@ -1570,15 +1977,182 @@ module ClaudeAgentSDK
1570
1977
  walk_to_leaf(by_uuid, uuid)
1571
1978
  end
1572
1979
 
1573
- # Keep only main-chain candidates (not sidechain, team, or meta)
1574
- main_leaves = leaf_candidates.reject do |e|
1575
- e['isSidechain'] || e['teamName'] || e['isMeta']
1980
+ best_leaf = pick_leaf(leaf_candidates, by_uuid, by_position)
1981
+ return [] unless best_leaf
1982
+
1983
+ reattach_parallel_tool_results(walk_to_root(by_uuid, best_leaf), entries, skip_flagged: true)
1984
+ end
1985
+
1986
+ # The leaf a conversation is read back from: the main-chain candidate
1987
+ # (not sidechain, team or meta) with the highest file position.
1988
+ #
1989
+ # Without one, fall back to the other candidates instead of reading the
1990
+ # conversation as empty, as Python does (`_pick_best(main_leaves) if
1991
+ # main_leaves else _pick_best(leaves)`): a session can end on a meta
1992
+ # entry nobody answered (a slash-command or skill body, a stop-hook
1993
+ # message, a system reminder — the user closed the session first), and
1994
+ # filter_visible_messages drops the flagged entries of the chain anyway.
1995
+ # Among those candidates one whose path to the root passes a visible
1996
+ # message comes first, then file position: the latest of them may be a
1997
+ # sidechain or teammate leaf with nothing visible above it, and taking
1998
+ # it would still read the conversation as empty.
1999
+ def pick_leaf(candidates, by_uuid, by_position)
2000
+ latest = ->(leaves) { leaves.max_by { |e| by_position[e['uuid']] || 0 } }
2001
+ main_leaves = candidates.reject { |e| off_main_conversation?(e) }
2002
+ return latest.call(main_leaves) unless main_leaves.empty?
2003
+
2004
+ known = {}
2005
+ with_visible = candidates.select { |e| visible_ancestor?(by_uuid, e, known) }
2006
+ latest.call(with_visible.empty? ? candidates : with_visible)
2007
+ end
2008
+
2009
+ # Whether the path from +leaf+ to its root passes an entry
2010
+ # filter_visible_messages returns. +known+ carries the answer for every
2011
+ # uuid already walked, so all the candidates of one transcript cost one
2012
+ # pass over it.
2013
+ def visible_ancestor?(by_uuid, leaf, known)
2014
+ walked = []
2015
+ current = leaf
2016
+ found = false
2017
+ while current && !known.key?(current['uuid'])
2018
+ known[current['uuid']] = false # a parentUuid cycle ends here
2019
+ walked << current['uuid']
2020
+ break if (found = visible_message?(current))
2021
+
2022
+ current = by_uuid[current['parentUuid']]
2023
+ end
2024
+ found ||= current ? known[current['uuid']] : false
2025
+ walked.each { |uuid| known[uuid] = found }
2026
+ found
2027
+ end
2028
+
2029
+ # An entry that is not part of the user's own conversation: written by a
2030
+ # subagent (sidechain) or a teammate, or a meta injection.
2031
+ def off_main_conversation?(entry)
2032
+ entry['isSidechain'] || entry['teamName'] || entry['isMeta']
2033
+ end
2034
+
2035
+ # A user/assistant entry of the user's own conversation.
2036
+ def visible_message?(entry)
2037
+ %w[user assistant].include?(entry['type']) && !off_main_conversation?(entry)
2038
+ end
2039
+
2040
+ # Put the results of parallel tool calls back on a leaf-to-root chain.
2041
+ #
2042
+ # The CLI writes one assistant entry per tool_use block (chained through
2043
+ # parentUuid) and parents every tool_result on the entry that holds ITS
2044
+ # tool_use. With two or more calls in one API message, only the result
2045
+ # the conversation continued from is an ancestor of the leaf; the others
2046
+ # are siblings of the next tool_use entry, and a single-path walk returns
2047
+ # their tool_use without them.
2048
+ #
2049
+ # For each assistant entry on the chain, take its user children that are
2050
+ # not on the chain and carry a tool_result for a tool_use on the chain,
2051
+ # and insert them — in file order — before the next user entry of the
2052
+ # chain (the batch's own on-chain result), or at the end when the chain
2053
+ # has none. The anchor, not the raw file position, decides the place: a
2054
+ # result that arrived after the conversation had already moved on to a
2055
+ # further tool_use of the same message still lands with its batch, so
2056
+ # every result follows the assistant turn that asked for it.
2057
+ #
2058
+ # The tool_use_id match is what keeps other user siblings out: a prompt
2059
+ # abandoned by a rewind is a second child of a chain entry too, and
2060
+ # starts a branch that was dropped — and so is the old result of a call
2061
+ # the conversation was rewound to and answered again: the chain's own
2062
+ # result for a tool_use_id wins, and at most one off-chain result per id
2063
+ # is ever added (the first in file order). Nor does a result that a user
2064
+ # or assistant entry went on from come back: after a rewind to the call
2065
+ # that continued with a new prompt, it heads the dropped branch.
2066
+ # +skip_flagged+ additionally
2067
+ # rejects sidechain / meta / team entries (main transcripts; a subagent
2068
+ # transcript is sidechain throughout).
2069
+ def reattach_parallel_tool_results(chain, entries, skip_flagged:)
2070
+ off_chain = off_chain_tool_results(chain, entries, skip_flagged)
2071
+ return chain if off_chain.empty?
2072
+
2073
+ placed = []
2074
+ pending = []
2075
+ chain.each do |entry|
2076
+ if entry['type'] == 'user' && !pending.empty?
2077
+ placed.concat(pending.sort_by(&:first).map(&:last))
2078
+ pending = []
2079
+ end
2080
+ placed << entry
2081
+ pending.concat(off_chain.fetch(entry['uuid'], []))
2082
+ end
2083
+ placed.concat(pending.sort_by(&:first).map(&:last))
2084
+ end
2085
+
2086
+ # { uuid of an assistant entry on the chain => [[file position, entry], ...] }
2087
+ # for the off-chain user children reattach_parallel_tool_results places.
2088
+ def off_chain_tool_results(chain, entries, skip_flagged) # rubocop:disable Metrics/AbcSize, Metrics/CyclomaticComplexity, Metrics/PerceivedComplexity -- one filter per condition of the re-attachment rule
2089
+ on_chain = Set.new
2090
+ assistants = Set.new
2091
+ tool_use_ids = Set.new
2092
+ answered = Set.new # tool_use ids the chain's own results answer
2093
+ chain.each do |entry|
2094
+ on_chain << entry['uuid']
2095
+ case entry['type']
2096
+ when 'assistant'
2097
+ assistants << entry['uuid']
2098
+ tool_use_ids.merge(content_block_values(entry, 'tool_use', 'id'))
2099
+ when 'user' then answered.merge(content_block_values(entry, 'tool_result', 'tool_use_id'))
2100
+ end
2101
+ end
2102
+ return {} if (tool_use_ids - answered).empty?
2103
+
2104
+ continued = continued_from(entries)
2105
+ found = {}
2106
+ entries.each_with_index do |entry, position|
2107
+ next unless entry['type'] == 'user' && assistants.include?(entry['parentUuid'])
2108
+ next if on_chain.include?(entry['uuid'])
2109
+ next if skip_flagged && off_main_conversation?(entry)
2110
+ # A result something went on from started a branch that a rewind
2111
+ # dropped (the conversation was rewound to the call and went on
2112
+ # with a new prompt). The results of parallel calls are never
2113
+ # continued from: the conversation goes on from the last one written.
2114
+ next if continued.include?(entry['uuid'])
2115
+
2116
+ ids = content_block_values(entry, 'tool_result', 'tool_use_id')
2117
+ next if ids.empty? || !ids.all? { |id| tool_use_ids.include?(id) && !answered.include?(id) }
2118
+
2119
+ # Claimed: a later result for the same call is not added — nor a
2120
+ # second copy of this entry, which a store can hold (a retried mirror
2121
+ # batch overlaps the write it retries).
2122
+ answered.merge(ids)
2123
+ (found[entry['parentUuid']] ||= []) << [position, entry]
2124
+ end
2125
+ found
2126
+ end
2127
+
2128
+ # uuids of the entries a user or assistant entry continues from: its
2129
+ # nearest user / assistant ancestor, reached through entries that are
2130
+ # neither (hook attachments, system entries) — so a hook attachment
2131
+ # written after an entry does not count as going on from it.
2132
+ def continued_from(entries)
2133
+ by_uuid = {}
2134
+ entries.each { |entry| by_uuid[entry['uuid']] = entry if entry['uuid'] }
2135
+ continued = Set.new
2136
+ entries.each do |entry|
2137
+ next unless %w[user assistant].include?(entry['type'])
2138
+
2139
+ seen = Set.new
2140
+ parent = by_uuid[entry['parentUuid']]
2141
+ parent = by_uuid[parent['parentUuid']] while parent && !%w[user assistant].include?(parent['type']) &&
2142
+ seen.add?(parent['uuid'])
2143
+ continued << parent['uuid'] if parent && %w[user assistant].include?(parent['type'])
1576
2144
  end
1577
- return [] if main_leaves.empty?
2145
+ continued
2146
+ end
1578
2147
 
1579
- # Pick the leaf with highest file position, walk to root
1580
- best_leaf = main_leaves.max_by { |e| by_position[e['uuid']] || 0 }
1581
- walk_to_root(by_uuid, best_leaf)
2148
+ # Values of +key+ over the +type+ content blocks of an entry's message
2149
+ # ([] for a message without array content — entries are opaque blobs).
2150
+ def content_block_values(entry, type, key)
2151
+ message = entry['message']
2152
+ content = message.is_a?(Hash) ? message['content'] : nil
2153
+ return [] unless content.is_a?(Array)
2154
+
2155
+ content.filter_map { |block| block[key] if block.is_a?(Hash) && block['type'] == type }
1582
2156
  end
1583
2157
 
1584
2158
  def walk_to_leaf(by_uuid, uuid)
@@ -1609,10 +2183,7 @@ module ClaudeAgentSDK
1609
2183
 
1610
2184
  def filter_visible_messages(chain)
1611
2185
  chain.filter_map do |entry|
1612
- next unless %w[user assistant].include?(entry['type'])
1613
- next if entry['isMeta']
1614
- next if entry['isSidechain']
1615
- next if entry['teamName']
2186
+ next unless visible_message?(entry)
1616
2187
 
1617
2188
  # NOTE: isCompactSummary messages are intentionally included. They contain
1618
2189
  # the summarized content from compacted conversations and are the only
@@ -1628,16 +2199,24 @@ module ClaudeAgentSDK
1628
2199
  end
1629
2200
  end
1630
2201
 
1631
- private_class_method :get_session_info_for_directory,
2202
+ private_class_method :absolute_path_keeping_dots, :resolve_missing_path, :symlink_target,
2203
+ :project_dir_records_cwd?, :prefix_fallback_dir?, :recorded_cwd,
2204
+ :each_parsed_entry, :get_session_info_for_directory,
1632
2205
  :list_sessions_for_directory, :list_all_sessions,
1633
2206
  :deduplicate_sessions, :dedup_rank,
1634
- :find_session_file, :stat_candidate, :resolve_subagents_dir,
1635
- :collect_agent_files, :parse_jsonl_entries,
2207
+ :worktree_paths, :find_session_file, :stat_candidate, :resolve_subagents_dir,
2208
+ :collect_agent_files, :parse_jsonl_entries, :utf8_transcript_text,
1636
2209
  :build_conversation_chain, :walk_to_leaf, :walk_to_root,
1637
- :filter_visible_messages, :read_head_tail, :build_session_info, :user_entry_texts,
2210
+ :pick_leaf, :visible_ancestor?, :off_main_conversation?, :visible_message?,
2211
+ :reattach_parallel_tool_results, :off_chain_tool_results,
2212
+ :content_block_values,
2213
+ :filter_visible_messages, :build_session_info, :created_at_from_file, :user_entry_texts,
2214
+ :each_line_past_head,
2215
+ :first_prompt_in, :first_prompt_from_file, :first_prompt_past_head,
1638
2216
  :valid_agent_id?, :sidechain_head?,
1639
2217
  :list_sessions_via_summaries, :paginate_resolving_gaps, :resolve_gap_slot,
1640
- :derive_info_from_entries, :mtime_from_entries, :apply_sort_limit_offset,
2218
+ :derive_info_from_entries, :store_session_mtime, :mtime_from_entries,
2219
+ :apply_sort_limit_offset,
1641
2220
  :filter_transcript_entries, :entries_to_messages,
1642
2221
  :entries_to_subagent_messages, :build_subagent_chain, :resolve_subagent_subpath,
1643
2222
  :import_subagent_files, :append_jsonl_file_in_batches, :collect_jsonl_files,
@@ -1646,6 +2225,9 @@ module ClaudeAgentSDK
1646
2225
  # These remain accessible for SessionMutations / SessionResume:
1647
2226
  # config_dir, sanitize_path, find_project_dir, detect_worktrees,
1648
2227
  # valid_session_id? (mutation boundary checks), listing_sort_key
1649
- # (--continue candidate order)
2228
+ # (--continue candidate order), nfc_path (SessionStores.projects_dir),
2229
+ # read_head_tail, title_and_first_prompt
2230
+ # and display_title (the fork title), own_transcript? (the mutations'
2231
+ # lookups)
1650
2232
  end
1651
2233
  end