capybara-simulated 0.3.0 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -13,6 +13,7 @@ require 'socket'
13
13
  require 'thread'
14
14
  require 'time'
15
15
  require 'uri'
16
+ require 'uri/idna' # WHATWG/UTS46 domain-to-ASCII/Unicode (uri-idna gem)
16
17
  require_relative 'asset_cache'
17
18
  require_relative 'errors'
18
19
  require_relative 'stack_resolver'
@@ -315,11 +316,23 @@ module Capybara
315
316
  # so long-running compute (e.g. mozjpeg over an 8900×8900 frame)
316
317
  # isn't starved by the settle_gen idle gate.
317
318
  @worker_in_flight = 0
319
+ # Workers whose initial script hasn't finished running yet. A worker that
320
+ # posts immediately on spawn (no main->worker message first) would leave
321
+ # `@worker_in_flight` at 0, so `worker_pending?` would be false in the gap
322
+ # between spawn and that first post — and settle / tick_real_time would
323
+ # stop waiting before the message lands. Count spawned-but-not-initialised
324
+ # workers so the async drain holds until the initial script has run.
325
+ @worker_initializing = 0
326
+ @worker_init_lock = Mutex.new
318
327
  # Cross-isolate `blob:` store. Worker isolates can't see the
319
328
  # main scope's `__csimBlobs` Map, so we mirror bytes here and
320
329
  # workers resolve them through a host fn.
321
330
  @blob_registry = {}
322
331
  @blob_registry_lock = Mutex.new
332
+ # url => owning worker handle, for blob URLs created INSIDE a worker. A
333
+ # worker's blob URL store dies with it, so terminating the worker revokes
334
+ # them (url-lifetime "Terminating worker revokes its URLs").
335
+ @blob_owners = {}
323
336
  # Postmessage transferable-buffer store. Large Uint8Array /
324
337
  # ArrayBuffer payloads cross isolates as a Ruby-side byte ID
325
338
  # rather than a JSON base64 string, so peak JS heap stays flat.
@@ -340,6 +353,9 @@ module Capybara
340
353
  # `message` event the next time it's active and settles/ticks. Plain
341
354
  # array (same thread — windows aren't background-threaded like workers).
342
355
  @window_inbox = []
356
+ # Cross-window BroadcastChannel messages from OTHER windows, delivered to
357
+ # this window's matching channels on settle. [{name, data}] (same thread).
358
+ @broadcast_inbox = []
343
359
  end
344
360
 
345
361
  # Worker thread polling and termination intervals — split so a
@@ -531,13 +547,14 @@ module Capybara
531
547
  # the find cache (its keys aren't realm-qualified, and a switch is rare).
532
548
  #
533
549
  # Scope: finds, reads, interactions (click/fill_in/…), evaluate_script,
534
- # and a self-targeted navigation (a link / form submit whose default
535
- # action loads a new document) all route into the frame — the frame's
536
- # realm is rebuilt from the fetched document, leaving the top page
537
- # untouched (see `navigate_frame`). Out of scope: `_top` navigates the
538
- # main page (correct), but a `_parent` target from a frame nested ≥2
539
- # levels navigates the main page rather than the intermediate frame, and
540
- # cross-origin frame locality is resolved against the main page's origin.
550
+ # and navigation (a link / form submit whose default action loads a new
551
+ # document) all route into the frame — the target frame's realm is rebuilt
552
+ # from the fetched document, leaving the top page untouched (see
553
+ # `navigate_frame` / `frame_nav_target_entry`). A `_parent`-targeted link
554
+ # or form from a frame nested ≥2 levels rebuilds the intermediate parent
555
+ # frame; `_top` (and a one-level `_parent`, whose parent is the top
556
+ # context) navigate the main page. Cross-origin frame locality is resolved
557
+ # against the main page's origin.
541
558
  def switch_to_frame(target)
542
559
  invalidate_find_cache
543
560
  case target
@@ -601,16 +618,27 @@ module Capybara
601
618
  end
602
619
 
603
620
  # Does a link/form `target` load into the CURRENT frame? Empty or `_self`
604
- # do; `_top` / `_blank` / `_parent` / a named context do not. `_top`
605
- # correctly navigates the main page (it falls through to `navigate`).
606
- # `_parent` from a frame nested ≥2 levels would ideally navigate the
607
- # intermediate parent frame, not the top page — that ancestor-targeted
608
- # case isn't modelled yet (rare); it currently navigates the main page.
621
+ # do; `_top` / `_blank` / `_parent` / a named context do not.
609
622
  def frame_self_target?(target)
610
623
  t = target.to_s.downcase
611
624
  t.empty? || t == '_self'
612
625
  end
613
626
 
627
+ # Resolve a link/form `target` to the frame stack entry its navigation
628
+ # should rebuild, or nil when it targets the top page / a new context
629
+ # (the caller then falls through to a full-page `navigate` or aux window).
630
+ # Only meaningful inside a frame (`@current_realm_id` set):
631
+ # - `''` / `_self` → the current frame.
632
+ # - `_parent` → the intermediate parent frame, but only when nested ≥2
633
+ # levels deep; at one level the parent IS the top browsing context, so
634
+ # it returns nil and the full-page path handles it (same as `_top`).
635
+ def frame_nav_target_entry(target)
636
+ return nil unless @current_realm_id
637
+ return @frame_stack.last if frame_self_target?(target)
638
+ return @frame_stack[-2] if target.to_s.downcase == '_parent' && @frame_stack.size >= 2
639
+ nil
640
+ end
641
+
614
642
  def find_css(css, context_handle = nil)
615
643
  s = css.to_s
616
644
  return find_xpath(s, context_handle) if xpath_shaped?(s)
@@ -869,19 +897,21 @@ module Capybara
869
897
  # downloads from `click_link` complete inside the click
870
898
  # action.
871
899
  consume_pending_location
900
+ consume_pending_frame_nav
872
901
  return
873
902
  end
874
903
  case action['kind']
875
904
  when 'navigate'
876
905
  url = action['url'].to_s
877
906
  target = action['target'].to_s
878
- # Inside a frame, a self-targeted link navigates the FRAME, not the
879
- # top page: fetch + rebuild this frame's realm. A pure-fragment link
880
- # is already handled in-realm by the frame's own location JS.
881
- if @current_realm_id && frame_self_target?(target)
882
- unless pure_fragment_navigation?(url)
907
+ # Inside a frame, a frame-targeted link (self, or `_parent` of a
908
+ # ≥2-deep frame) navigates that FRAME, not the top page: fetch +
909
+ # rebuild its realm. A self-targeted pure-fragment link is already
910
+ # handled in-realm by the frame's own location JS, so skip it.
911
+ if (frame_entry = frame_nav_target_entry(target))
912
+ unless frame_entry.equal?(@frame_stack.last) && pure_fragment_navigation?(url)
883
913
  tick_real_time
884
- navigate_frame(resolve_against_current(url, use_base: true))
914
+ navigate_frame(resolve_against_current(url, use_base: true), entry: frame_entry)
885
915
  end
886
916
  # `target="_blank"` (or any non-_self/_top/_parent name) opens
887
917
  # in a new browsing context (its own Browser/VM); the primary
@@ -890,7 +920,7 @@ module Capybara
890
920
  # to `noopener` (so `window.opener` is null), unlike JS `window.open`
891
921
  # which keeps the opener — see `open_window_from_js`.
892
922
  elsif !target.empty? && !%w[_self _top _parent].include?(target.downcase) && @driver.respond_to?(:open_aux_window)
893
- @driver.open_aux_window(resolve_against_current(url, use_base: true))
923
+ @driver.open_aux_window(resolve_against_current(url, use_base: true), source: self, blob_snapshot: action['blob'])
894
924
  # In-page anchor links (`#frag` / current-page + `#frag`) move
895
925
  # the hash but don't fetch a new document. Pure-fragment also
896
926
  # short-circuits the `<a>`s test fixtures use as click sinks.
@@ -930,6 +960,7 @@ module Capybara
930
960
  # pending; if `@current_url` already changed mid-drain (the
931
961
  # navigate landed during a timer fire), skip the form submit
932
962
  # entirely — its form handle is in a stale VM by now.
963
+ consume_pending_frame_nav
933
964
  if @pending_location
934
965
  consume_pending_location
935
966
  elsif @current_url != submit_baseline_url
@@ -1444,7 +1475,7 @@ module Capybara
1444
1475
  url = pending['url'].to_s
1445
1476
  target = pending['target'].to_s
1446
1477
  if !target.empty? && !%w[_self _top _parent].include?(target.downcase) && @driver.respond_to?(:open_aux_window)
1447
- @driver.open_aux_window(resolve_against_current(url, use_base: true))
1478
+ @driver.open_aux_window(resolve_against_current(url, use_base: true), source: self, blob_snapshot: pending['blob'])
1448
1479
  elsif pure_fragment_navigation?(url)
1449
1480
  update_current_hash(url)
1450
1481
  else
@@ -1773,7 +1804,7 @@ module Capybara
1773
1804
  @stack_resolver ||= StackResolver.new(self)
1774
1805
  end
1775
1806
 
1776
- def log_network(method, url, status) = @trace&.log_network(method, url, status)
1807
+ def log_network(method, url, status, **extra) = @trace&.log_network(method, url, status, **extra)
1777
1808
 
1778
1809
  # `tag#id.class` short description of the handle, for trace
1779
1810
  # `description` fields. One V8 round-trip; only paid when a step
@@ -2105,15 +2136,15 @@ module Capybara
2105
2136
  URI.encode_www_form(fields)
2106
2137
  end
2107
2138
  action_url = action.empty? ? (current_browsing_context_url || @default_host) : resolve_against_current(action)
2108
- # A form submitted inside a frame whose target is the frame itself
2109
- # navigates the FRAME, not the top page.
2110
- in_frame = !!@current_realm_id && frame_self_target?(spec['target'])
2139
+ # A form submitted inside a frame whose target is that frame (self, or a
2140
+ # `_parent` of a ≥2-deep frame) navigates the FRAME, not the top page.
2141
+ frame_entry = frame_nav_target_entry(spec['target'])
2111
2142
  if method == 'GET'
2112
2143
  uri = URI.parse(action_url)
2113
2144
  uri.query = body unless body.empty?
2114
- in_frame ? navigate_frame(uri.to_s) : navigate(uri.to_s)
2115
- elsif in_frame
2116
- navigate_frame_post(action_url, body, content_type || enctype)
2145
+ frame_entry ? navigate_frame(uri.to_s, entry: frame_entry) : navigate(uri.to_s)
2146
+ elsif frame_entry
2147
+ navigate_frame_post(action_url, body, content_type || enctype, entry: frame_entry)
2117
2148
  else
2118
2149
  navigate_post(action_url, body, content_type || enctype)
2119
2150
  end
@@ -2126,7 +2157,7 @@ module Capybara
2126
2157
  append_multipart_part(body, boundary, name, value.to_s)
2127
2158
  end
2128
2159
  file_inputs.each do |fi|
2129
- picks = @file_picks && @file_picks[fi['handle'].to_i] || []
2160
+ picks = file_pick_paths(fi)
2130
2161
  if picks.empty?
2131
2162
  append_multipart_part(body, boundary, fi['name'].to_s, '', filename: '')
2132
2163
  else
@@ -2141,6 +2172,34 @@ module Capybara
2141
2172
  {content_type: "multipart/form-data; boundary=#{boundary}", body: body}
2142
2173
  end
2143
2174
 
2175
+ # The on-disk paths backing a file input's current selection. Each
2176
+ # selected File reports its host-backed source (`handle`/`index` → the
2177
+ # `@file_picks` slot recorded at `attach_file` time); this resolves bytes
2178
+ # even when JS moved a File onto a different input (`input.files =
2179
+ # dataTransfer.files`), whose own handle was never attached to. Falls back
2180
+ # to the input's own handle for older serializer payloads.
2181
+ #
2182
+ # Only host-backed Files (from `attach_file`) resolve here; a purely
2183
+ # in-memory `new File(['bytes'], …)` assigned via JS has no `@file_picks`
2184
+ # slot, so a CLASSIC (non-Turbo) submit drops its bytes — the fetch/XHR
2185
+ # path serializes those in JS (`serializeMultipart` → `blobBytes`) and is
2186
+ # unaffected. This matches the pre-existing behaviour and covers every
2187
+ # realistic upload (host-backed file submitted through Turbo or a plain
2188
+ # form).
2189
+ def file_pick_paths(fi)
2190
+ refs = fi['files']
2191
+ if refs.is_a?(Array) && !refs.empty?
2192
+ refs.filter_map {|ref|
2193
+ handle = ref['handle']
2194
+ next if handle.nil?
2195
+ picks = @file_picks && @file_picks[handle.to_i]
2196
+ picks && picks[ref['index'].to_i]
2197
+ }
2198
+ else
2199
+ (@file_picks && @file_picks[fi['handle'].to_i]) || []
2200
+ end
2201
+ end
2202
+
2144
2203
  def append_multipart_part(body, boundary, name, content, filename: nil, content_type: nil)
2145
2204
  body << "--#{boundary}\r\n"
2146
2205
  disposition = %[form-data; name="#{name}"]
@@ -2236,10 +2295,11 @@ module Capybara
2236
2295
  reset_workers
2237
2296
  reset_websockets
2238
2297
  @window_inbox.clear
2298
+ @broadcast_inbox.clear
2239
2299
  # Free any zero-copy transfer backing stores that went unimported
2240
2300
  # (worker killed before draining its inbox, etc.) before the rebuild.
2241
2301
  drop_pending_transfers
2242
- @blob_registry_lock.synchronize { @blob_registry.clear }
2302
+ @blob_registry_lock.synchronize { @blob_registry.clear; @blob_owners.clear }
2243
2303
  # Drop volatile entries from the class-level HTTP asset cache
2244
2304
  # so test-local DB state (TranslationOverride, etc.) reaches
2245
2305
  # the app on subsequent visits. Fingerprinted assets
@@ -2268,6 +2328,14 @@ module Capybara
2268
2328
  reset_hijacked_fetches
2269
2329
  reset_websockets
2270
2330
  @window_inbox.clear
2331
+ @broadcast_inbox.clear
2332
+ # Dispose the JS runtime/isolate itself — for an auxiliary window this
2333
+ # Browser is the isolate's last owner, but V8Runtime registers every
2334
+ # isolate in a process-wide `@@live` set (for at_exit cleanup), which
2335
+ # pins it past a bare GC. Without this, each closed window leaked a live
2336
+ # V8 isolate (RSS climbed across a long suite). Only reached on teardown,
2337
+ # never on the per-test `reset!` path (which keeps the runtime).
2338
+ @runtime.dispose if @runtime.respond_to?(:dispose)
2271
2339
  rescue StandardError
2272
2340
  nil
2273
2341
  end
@@ -2615,10 +2683,16 @@ module Capybara
2615
2683
 
2616
2684
  # `binary` is set by the JS side (it knows whether `send` was given a
2617
2685
  # string or an ArrayBuffer/view) → opcode 0x2 vs the text 0x1. Action
2618
- # Cable is text-only (JSON). The payload's bytes are written as-is.
2619
- def ws_send(id, data, binary = false)
2686
+ # Cable is text-only (JSON). `b64` is set when the bytes arrived base64-
2687
+ # encoded (the QuickJS binary path — raw bytes ≥0x80 don't survive its
2688
+ # host boundary); decode before framing.
2689
+ def ws_send(id, data, binary = false, b64 = false)
2620
2690
  sock = @websocket_sockets[id.to_i] or return
2621
- ws_write_frame(sock, binary ? 0x2 : 0x1, data.to_s.b)
2691
+ if binary
2692
+ ws_write_frame(sock, 0x2, b64 ? Base64.decode64(data.to_s) : data.to_s.b)
2693
+ else
2694
+ ws_write_frame(sock, 0x1, data.to_s.b)
2695
+ end
2622
2696
  nil
2623
2697
  rescue StandardError
2624
2698
  nil
@@ -2990,7 +3064,7 @@ module Capybara
2990
3064
  # worker's `__csim_workerPostMessage` host fn closes over its
2991
3065
  # handle and routes outgoing messages onto a shared outbox the
2992
3066
  # main settle drains.
2993
- def worker_spawn(url)
3067
+ def worker_spawn(url, shared: false)
2994
3068
  handle = (@worker_seq += 1)
2995
3069
  inbox = Thread::Queue.new
2996
3070
  outbox = @worker_outbox
@@ -3002,9 +3076,11 @@ module Capybara
3002
3076
  # non-owning thread SEGVs (V8 isolates are thread-
3003
3077
  # bound; quickjs.rb's VM is similarly per-thread).
3004
3078
  body = fetch_worker_script(target)
3079
+ # Pending until the worker's initial script has run (see @worker_initializing).
3080
+ @worker_init_lock.synchronize { @worker_initializing += 1 }
3005
3081
  thread = Thread.new do
3006
3082
  Thread.current.report_on_exception = false
3007
- run_worker(handle, target, body, inbox, outbox, engine_class)
3083
+ run_worker(handle, target, body, inbox, outbox, engine_class, shared: shared)
3008
3084
  end
3009
3085
  @workers[handle] = {thread: thread, inbox: inbox}
3010
3086
  handle
@@ -3022,13 +3098,21 @@ module Capybara
3022
3098
  return unless w
3023
3099
  w[:inbox] << :terminate
3024
3100
  # Most clean shutdowns are <10 ms; the kill is the fallback
3025
- # for blocked workers.
3101
+ # for blocked workers. Join again AFTER the kill so the thread is actually
3102
+ # dead before we revoke its URLs — `Thread#kill` is async, and a worker
3103
+ # still running a `createObjectURL` could otherwise re-register a URL after
3104
+ # the revoke and leak it.
3026
3105
  w[:thread].join(WORKER_TERMINATE_GRACE)
3027
- w[:thread].kill if w[:thread].alive?
3106
+ if w[:thread].alive?
3107
+ w[:thread].kill
3108
+ w[:thread].join(WORKER_TERMINATE_GRACE)
3109
+ end
3028
3110
  # A blocked worker that never returned messages leaves
3029
3111
  # `@worker_in_flight` permanently > 0; reset when no workers
3030
3112
  # remain so `polling?` can short-circuit again.
3031
3113
  @worker_in_flight = 0 if @workers.empty?
3114
+ # The worker is gone — revoke the blob URLs it created.
3115
+ revoke_worker_blobs(handle.to_i)
3032
3116
  end
3033
3117
 
3034
3118
  def deliver_worker_messages
@@ -3042,7 +3126,7 @@ module Capybara
3042
3126
  events.size
3043
3127
  end
3044
3128
 
3045
- def worker_pending? = !@worker_outbox.empty? || @worker_in_flight > 0
3129
+ def worker_pending? = !@worker_outbox.empty? || @worker_in_flight > 0 || @worker_init_lock.synchronize { @worker_initializing } > 0
3046
3130
 
3047
3131
  # ── Cross-window messaging (window.open / opener / postMessage) ──
3048
3132
  # Each window is a separate Browser/VM/isolate, so a reference to another
@@ -3066,6 +3150,17 @@ module Capybara
3066
3150
  end
3067
3151
 
3068
3152
  def window_location_of(handle) = @driver.respond_to?(:window_location) ? @driver.window_location(handle.to_s).to_s : ''
3153
+ # Cross-window property reads (a WindowProxy `win.foo` / `win.document.foo`):
3154
+ # route to the Driver, which reads a PRIMITIVE off the target window's VM.
3155
+ def window_get(handle, prop) = (@driver.respond_to?(:window_read) ? @driver.window_read(handle.to_s, prop.to_s, doc: false) : nil)
3156
+ def window_doc_get(handle, prop) = (@driver.respond_to?(:window_read) ? @driver.window_read(handle.to_s, prop.to_s, doc: true) : nil)
3157
+ # Read a primitive property off THIS window's globalThis / document — called
3158
+ # by the Driver to serve another window's cross-window proxy read.
3159
+ def read_property(prop, doc: false)
3160
+ @runtime.call('__csimReadWindowProp', doc, prop.to_s)
3161
+ rescue StandardError
3162
+ nil
3163
+ end
3069
3164
  def set_window_location(handle, url) = (@driver.window_set_location(handle.to_s, url.to_s) if @driver.respond_to?(:window_set_location))
3070
3165
  def window_closed?(handle) = @driver.respond_to?(:window_closed?) ? @driver.window_closed?(handle.to_s) : true
3071
3166
  def close_child_window(handle) = (@driver.close_window(handle.to_s) if @driver.respond_to?(:close_window))
@@ -3078,14 +3173,34 @@ module Capybara
3078
3173
  @window_inbox << {'data' => data, 'origin' => origin.to_s, 'sourceHandle' => source_handle.to_s}
3079
3174
  end
3080
3175
 
3081
- def window_message_pending? = !@window_inbox.empty?
3176
+ # Covers both cross-window postMessage AND BroadcastChannel — the two
3177
+ # cross-window event channels share these drain/pending hooks.
3178
+ def window_message_pending? = !@window_inbox.empty? || !@broadcast_inbox.empty?
3179
+
3180
+ # A BroadcastChannel message from another window, queued for delivery to
3181
+ # this window's channels with the same name.
3182
+ def enqueue_broadcast(name, data) = (@broadcast_inbox << {'name' => name.to_s, 'data' => data})
3082
3183
 
3083
- # Fire queued cross-window messages as `message` events on window.
3184
+ # Fire queued cross-window messages (postMessage + BroadcastChannel).
3084
3185
  def deliver_window_messages
3085
- return 0 if @window_inbox.empty?
3086
- events = @window_inbox.slice!(0, @window_inbox.length)
3087
- @runtime.call('__csim_deliverWindowMessages', events)
3088
- events.size
3186
+ n = 0
3187
+ unless @window_inbox.empty?
3188
+ events = @window_inbox.slice!(0, @window_inbox.length)
3189
+ @runtime.call('__csim_deliverWindowMessages', events)
3190
+ n += events.size
3191
+ end
3192
+ unless @broadcast_inbox.empty?
3193
+ events = @broadcast_inbox.slice!(0, @broadcast_inbox.length)
3194
+ @runtime.call('__csim_deliverBroadcasts', events)
3195
+ n += events.size
3196
+ end
3197
+ n
3198
+ end
3199
+
3200
+ # `BroadcastChannel.postMessage` in THIS window — fan out to every OTHER
3201
+ # window's matching channels (same-window delivery happens in-VM).
3202
+ def broadcast_to_windows(name, data)
3203
+ @driver.broadcast_channel(self, name.to_s, data) if @driver.respond_to?(:broadcast_channel)
3089
3204
  end
3090
3205
 
3091
3206
  # ── Image decode (libvips) ─────────────────────────────────────
@@ -3148,8 +3263,24 @@ module Capybara
3148
3263
  }
3149
3264
  end
3150
3265
 
3151
- def blob_register(url, body_b64)
3152
- @blob_registry_lock.synchronize { @blob_registry[url.to_s] = body_b64.to_s }
3266
+ def blob_register(url, body_b64, owner_realm = nil)
3267
+ # Tag the creating context so the URL is revoked when that context goes
3268
+ # away: a WORKER (separate thread, tagged via Thread.current) when it
3269
+ # terminates ("Terminating worker"), or a FRAME REALM (owner_realm passed
3270
+ # from JS createObjectURL) when the iframe is removed ("Removing an
3271
+ # iframe"). Namespaced ('w:' / 'r:') so a worker handle and a realm id
3272
+ # never collide. Main-realm blobs (no owner) live until clear_volatile.
3273
+ worker = Thread.current[:csim_worker_handle]
3274
+ key = if worker then "w:#{worker}"
3275
+ elsif owner_realm && owner_realm.to_i != 0 then "r:#{owner_realm.to_i}"
3276
+ end
3277
+ @blob_registry_lock.synchronize do
3278
+ @blob_registry[url.to_s] = body_b64.to_s
3279
+ # Keep ownership in sync both ways: a (re-)registration with no owner
3280
+ # (main thread / main realm) must DROP any prior owner, else revoking
3281
+ # that context would wrongly revoke a now-page-owned URL.
3282
+ if key then @blob_owners[url.to_s] = key else @blob_owners.delete(url.to_s) end
3283
+ end
3153
3284
  nil
3154
3285
  end
3155
3286
 
@@ -3157,11 +3288,73 @@ module Capybara
3157
3288
  @blob_registry_lock.synchronize { @blob_registry[url.to_s] }
3158
3289
  end
3159
3290
 
3291
+ # WHATWG URL "domain to ASCII" — the JS tr46 stub delegates non-ASCII / xn--
3292
+ # hosts here (the ASCII fast path stays in-VM). Returns the punycode form, or
3293
+ # nil on an IDNA failure (so whatwg-url reports "domain to ASCII failed").
3294
+ # `be_strict: false` is the URL parser's mode (UseSTD3ASCIIRules and
3295
+ # VerifyDnsLength off) — empty middle labels (`x..y`) and `_`/etc. are
3296
+ # allowed, matching whatwg-url's `domainToASCII(domain, false)`.
3297
+ def domain_to_ascii(domain)
3298
+ URI::IDNA.whatwg_to_ascii(domain.to_s, be_strict: false)
3299
+ rescue URI::IDNA::Error
3300
+ nil # a genuine IDNA failure (bad punycode / disallowed codepoint) — let
3301
+ # whatwg-url report "domain to ASCII failed". Non-IDNA errors propagate.
3302
+ end
3303
+
3304
+ # WHATWG URL "domain to Unicode" — best-effort (never fails the parse per
3305
+ # spec), so on an IDNA error fall back to the input domain (unlike to_ascii,
3306
+ # which signals failure with nil — the asymmetry is intentional).
3307
+ def domain_to_unicode(domain)
3308
+ URI::IDNA.whatwg_to_unicode(domain.to_s, be_strict: false)
3309
+ rescue URI::IDNA::Error
3310
+ domain.to_s
3311
+ end
3312
+
3313
+ # Read a blob URL's bytes + content type from THIS window's VM (its local
3314
+ # blob store) — the Driver uses it to load a blob: document into a fresh aux
3315
+ # window opened by this window. Returns {bytes:, type:} or nil.
3316
+ def read_blob_for_window(url)
3317
+ r = @runtime.call('__csimReadBlobForWindow', url.to_s)
3318
+ return nil unless r.is_a?(Hash) && r['b64']
3319
+ { bytes: Base64.decode64(r['b64'].to_s), type: r['type'].to_s }
3320
+ rescue StandardError
3321
+ nil
3322
+ end
3323
+
3324
+ # Load a blob: document (bytes from the opener) as THIS window's top-level
3325
+ # document — for `window.open(blobURL)` / a blob: aux-window navigation,
3326
+ # where the blob isn't rack-navigable and lives in the opener's isolate.
3327
+ def boot_blob_document(url, bytes, content_type)
3328
+ @current_url = url.to_s
3329
+ ct = content_type.to_s.empty? ? 'text/html' : content_type.to_s
3330
+ # Blob string parts are UTF-8-encoded; when the Blob type carries no
3331
+ # charset, decode the document as UTF-8 (not the windows-1252 HTML locale
3332
+ # default, which is an HTTP concept that doesn't apply to in-memory blobs —
3333
+ # matches the iframe blob: path's decodeBlobBody). A charset in the Blob
3334
+ # type (url-charset) is preserved so it can override <meta charset>.
3335
+ ct = "#{ct};charset=utf-8" unless ct.downcase.include?('charset')
3336
+ record_response(200, {'content-type' => ct})
3337
+ boot_response_into_ctx(bytes)
3338
+ end
3339
+
3160
3340
  def blob_unregister(url)
3161
- @blob_registry_lock.synchronize { @blob_registry.delete(url.to_s) }
3341
+ @blob_registry_lock.synchronize { @blob_registry.delete(url.to_s); @blob_owners.delete(url.to_s) }
3162
3342
  nil
3163
3343
  end
3164
3344
 
3345
+ # Revoke every blob URL owned by a context that's going away (its blob URL
3346
+ # store is part of the global being torn down).
3347
+ def revoke_owned_blobs(key)
3348
+ @blob_registry_lock.synchronize do
3349
+ urls = @blob_owners.select {|_url, owner| owner == key }.keys
3350
+ urls.each {|url| @blob_registry.delete(url); @blob_owners.delete(url) }
3351
+ end
3352
+ end
3353
+ # Keys are normalized with `.to_i` on BOTH sides (register tags
3354
+ # "r:#{owner_realm.to_i}") so a marshalled Float/String id still matches.
3355
+ def revoke_worker_blobs(handle) = revoke_owned_blobs("w:#{handle.to_i}")
3356
+ def revoke_realm_blobs(realm_id) = revoke_owned_blobs("r:#{realm_id.to_i}")
3357
+
3165
3358
  # ── postMessage transferable-buffer registry ───────────────────
3166
3359
  #
3167
3360
  # Large Uint8Array / ArrayBuffer payloads cross isolates by ID;
@@ -3300,7 +3493,20 @@ module Capybara
3300
3493
  # `build_worker` factory, evaluates the worker script, then
3301
3494
  # loops draining microtasks + timers + inbox until `:terminate`
3302
3495
  # lands or an exception propagates.
3303
- private def run_worker(handle, url, body, inbox, outbox, engine_class)
3496
+ private def run_worker(handle, url, body, inbox, outbox, engine_class, shared: false)
3497
+ # Release the spawn-time `@worker_initializing` count exactly once, however
3498
+ # this method exits (normal start, `self.close()`, or an exception), so
3499
+ # worker_pending? doesn't stay stuck true forever.
3500
+ initializing = true
3501
+ release_init = lambda do
3502
+ next unless initializing
3503
+ initializing = false
3504
+ @worker_init_lock.synchronize { @worker_initializing -= 1 }
3505
+ end
3506
+ # Tag this thread so blob URLs created by the worker's script are owned by
3507
+ # this handle and revoked on terminate (see blob_register / revoke_worker_blobs).
3508
+ Thread.current[:csim_worker_handle] = handle
3509
+ rt = nil
3304
3510
  raise "worker script not found: #{url}" unless body
3305
3511
  # The worker SCRIPT is text; the Rack-fetched body arrives
3306
3512
  # BINARY-tagged (see `RuntimeShared.utf8_text`).
@@ -3313,16 +3519,34 @@ module Capybara
3313
3519
  # the snapshot-time `http://placeholder/`.
3314
3520
  rt.eval("globalThis.__csimUpdateLocation(#{JSON.generate(url.to_s)});")
3315
3521
  rt.eval(body)
3316
- loop do
3317
- msg = pop_with_timeout(inbox, WORKER_POLL_INTERVAL)
3318
- break if msg == :terminate
3319
- rt.call('__csim_workerOnMessage', msg) if msg
3522
+ rt.drain_microtasks
3523
+ # A SharedWorker fires `connect` AFTER its script set `self.onconnect`; the
3524
+ # connect handler's port post lands in the outbox before release_init, so
3525
+ # worker_pending? stays true until it's delivered.
3526
+ if shared
3527
+ rt.eval('typeof __csimFireSharedWorkerConnect === "function" && __csimFireSharedWorkerConnect();')
3320
3528
  rt.drain_microtasks
3321
- rt.drain_timers if rt.has_ready_timer?
3529
+ end
3530
+ # Initial script has run (and any immediate postMessage is in the outbox).
3531
+ release_init.call
3532
+ # A worker that called `self.close()` in its top-level script stops here —
3533
+ # the script ran (and may have posted), but no further messages are pulled.
3534
+ unless rt.eval('!!globalThis.__csimWorkerClosed')
3535
+ loop do
3536
+ msg = pop_with_timeout(inbox, WORKER_POLL_INTERVAL)
3537
+ break if msg == :terminate
3538
+ if msg
3539
+ rt.call('__csim_workerOnMessage', msg)
3540
+ rt.drain_microtasks
3541
+ rt.drain_timers if rt.has_ready_timer?
3542
+ break if rt.eval('!!globalThis.__csimWorkerClosed')
3543
+ end
3544
+ end
3322
3545
  end
3323
3546
  rescue StandardError => e
3324
3547
  outbox << {handle: handle, kind: '__error', message: "#{e.class}: #{e.message}"}
3325
3548
  ensure
3549
+ release_init.call # guarantee the init count is released on an early raise
3326
3550
  rt&.dispose
3327
3551
  end
3328
3552
 
@@ -3332,10 +3556,31 @@ module Capybara
3332
3556
  # circuit to the JS-side blob registry instead. Http(s) URLs
3333
3557
  # fall through to the regular Rack path.
3334
3558
  private def fetch_worker_script(url)
3335
- return rack_fetch_body(url) unless url.to_s.start_with?('blob:')
3336
- b64 = @runtime.call('__csimReadBlobBase64', url)
3337
- return nil unless b64
3338
- Base64.decode64(b64.to_s)
3559
+ u = url.to_s
3560
+ if u.start_with?('blob:')
3561
+ b64 = @runtime.call('__csimReadBlobBase64', u)
3562
+ return nil unless b64
3563
+ return Base64.decode64(b64.to_s)
3564
+ end
3565
+ # `data:[<mediatype>][;base64],<data>` worker scripts (a worker created
3566
+ # from a data: URL — its origin is opaque, so its blob: URLs serialize
3567
+ # with a 'null' origin). Decode inline; Rack can't serve a data: URL.
3568
+ return decode_data_url_body(u) if u.start_with?('data:')
3569
+ rack_fetch_body(u)
3570
+ end
3571
+
3572
+ # The decoded body of a `data:[<mediatype>][;base64],<data>` URL (RFC 2397):
3573
+ # base64-decoded when the `;base64` flag is present, else percent-decoded.
3574
+ private def decode_data_url_body(url)
3575
+ comma = url.index(',')
3576
+ return '' unless comma
3577
+ meta = url[5...comma]
3578
+ payload = url[(comma + 1)..]
3579
+ if meta =~ /;base64\s*\z/i
3580
+ Base64.decode64(payload)
3581
+ else
3582
+ CGI.unescape(payload)
3583
+ end
3339
3584
  end
3340
3585
 
3341
3586
  # `Thread::Queue#pop(timeout:)` blocks releasing the GVL — fine
@@ -3420,12 +3665,14 @@ module Capybara
3420
3665
  headers = headers.reject {|k, _| k == 'X-Csim-Body-B64' }
3421
3666
  end
3422
3667
  MAX_FETCH_REDIRECTS.times do
3668
+ t0 = @trace && Process.clock_gettime(Process::CLOCK_MONOTONIC)
3423
3669
  # GET-only cache shortcut (RFC 9111). Fresh hit → skip @app.call
3424
3670
  # entirely; stale-but-revalidatable → fall through with conditional
3425
3671
  # headers added so the server can return 304.
3426
3672
  cache_entry = method == 'GET' ? @@asset_cache.lookup(target) : nil
3427
3673
  if cache_entry&.fresh?
3428
- log_network(method, target, cache_entry.status)
3674
+ # Cached static asset — log headers/type/size but skip the (boring) body.
3675
+ trace_network(method, target, cache_entry.status, headers, body, cache_entry.headers, nil, t0, false)
3429
3676
  return response_hash(cache_entry.status, cache_entry.headers, cache_entry.body, target, redirected)
3430
3677
  end
3431
3678
 
@@ -3436,14 +3683,16 @@ module Capybara
3436
3683
  env.merge!(env_extras) if env_extras
3437
3684
  status, resp_headers, resp_body = dispatch_rack_or_http(target, env, method: method, body: body)
3438
3685
  merge_set_cookie(resp_headers)
3439
- log_network(method, target, status)
3440
3686
  if status == 304 && cache_entry
3687
+ trace_network(method, target, cache_entry.status, headers, body, cache_entry.headers, nil, t0, false)
3441
3688
  resp_body.close if resp_body.respond_to?(:close)
3442
3689
  @@asset_cache.refresh(cache_entry, resp_headers)
3443
3690
  return response_hash(cache_entry.status, cache_entry.headers, cache_entry.body, target, redirected)
3444
3691
  end
3445
3692
  if redirect_mode != 'manual' && (loc = redirect_location(status, resp_headers))
3446
3693
  raise StandardError, '[capybara-simulated] fetch: redirect blocked by redirect=error mode' if redirect_mode == 'error'
3694
+ # Log this hop (3xx) before method/body are rewritten for the next.
3695
+ trace_network(method, target, status, headers, body, resp_headers, nil, t0, true)
3447
3696
  redirected = true
3448
3697
  preserve = [307, 308].include?(status)
3449
3698
  next_url = resolve_against(loc, target)
@@ -3454,6 +3703,7 @@ module Capybara
3454
3703
  next
3455
3704
  end
3456
3705
  body_str = read_rack_body(resp_body)
3706
+ trace_network(method, target, status, headers, body, resp_headers, body_str, t0, false)
3457
3707
  @@asset_cache.store(target, status, resp_headers, body_str) if method == 'GET'
3458
3708
  return response_hash(status, resp_headers, body_str, target, redirected)
3459
3709
  end
@@ -3463,6 +3713,63 @@ module Capybara
3463
3713
  nil
3464
3714
  end
3465
3715
 
3716
+ # Cap per-body capture so one big asset/response can't bloat the
3717
+ # trace. Generous (this is a local debugging artifact).
3718
+ NETWORK_BODY_CAP = 256 * 1024
3719
+
3720
+ # Enriched network log for the trace: response content-type / byte
3721
+ # size / elapsed ms / redirect flag, plus request + response headers
3722
+ # and bodies (devtools-style). No-ops — and skips all the lookups —
3723
+ # unless a trace is recording, so the fetch hot path is unaffected
3724
+ # when tracing is off.
3725
+ def trace_network(method, url, status, req_headers, req_body, resp_headers, resp_body, t0, redirected)
3726
+ return unless @trace
3727
+ ct = resp_headers && (resp_headers['content-type'] || resp_headers['Content-Type'])
3728
+ ct = ct.first if ct.is_a?(Array) # Rack 3 permits array-valued header fields
3729
+ ct = ct.split(';', 2).first.strip if ct.is_a?(String)
3730
+ size = if resp_body
3731
+ resp_body.bytesize
3732
+ elsif (cl = resp_headers && (resp_headers['content-length'] || resp_headers['Content-Length']))
3733
+ (cl.is_a?(Array) ? cl.first : cl).to_i
3734
+ end
3735
+ log_network(method, url, status,
3736
+ content_type: (ct if ct.is_a?(String)),
3737
+ size: size,
3738
+ duration_ms: (t0 && ((Process.clock_gettime(Process::CLOCK_MONOTONIC) - t0) * 1000).round),
3739
+ redirected: (redirected || nil),
3740
+ request_headers: normalize_trace_headers(req_headers),
3741
+ request_body: (req_body && !req_body.to_s.empty? ? cap_trace_body(req_body) : nil),
3742
+ response_headers: normalize_trace_headers(resp_headers),
3743
+ response_body: (resp_body ? cap_trace_body(resp_body) : nil))
3744
+ rescue StandardError => e
3745
+ # A trace-logging bug must NEVER break the real fetch: rack_fetch's
3746
+ # own `rescue StandardError` would otherwise swallow it and return
3747
+ # nil, so the asset (e.g. jQuery) silently fails to load. Drop the
3748
+ # log entry instead.
3749
+ warn "capybara-simulated: trace network log failed: #{e.class}: #{e.message}"
3750
+ end
3751
+
3752
+ # JSON-safe body for the trace: binary (non-UTF-8) bodies become a
3753
+ # placeholder rather than mojibake, and long bodies are truncated
3754
+ # (scrubbed so a mid-codepoint cut can't yield invalid UTF-8).
3755
+ #
3756
+ # Rack response bodies are ASCII-8BIT (BINARY); reinterpret the bytes as
3757
+ # UTF-8 up front and keep working in UTF-8 throughout. Otherwise a body
3758
+ # whose bytes ARE valid UTF-8 but stays BINARY-tagged would flow out of
3759
+ # here still BINARY, and the first concat with a UTF-8 string (the
3760
+ # truncation marker here, or the trace-buffer / JSON serialization
3761
+ # downstream) raises Encoding::CompatibilityError on any byte ≥ 0x80.
3762
+ def cap_trace_body(body)
3763
+ s = body.to_s.dup.force_encoding('UTF-8')
3764
+ return "[binary, #{s.bytesize} bytes]" unless s.valid_encoding?
3765
+ s.bytesize > NETWORK_BODY_CAP ? (s.byteslice(0, NETWORK_BODY_CAP).scrub + "\n…[truncated, #{s.bytesize} bytes total]") : s
3766
+ end
3767
+
3768
+ def normalize_trace_headers(headers)
3769
+ return nil unless headers
3770
+ headers.each_with_object({}) {|(k, v), out| out[k.to_s] = v.is_a?(Array) ? v.join(', ') : v.to_s }
3771
+ end
3772
+
3466
3773
  # CGI convention: `Content-Type` and `Content-Length` land in env
3467
3774
  # *without* the HTTP_ prefix. Rails / Rack params parsing reads
3468
3775
  # `CONTENT_TYPE` and dispatches JSON / multipart parsers off it;
@@ -3499,7 +3806,14 @@ module Capybara
3499
3806
  # encoding per the HTML "decode" algorithm and is removed). The real
3500
3807
  # bytes for binary consumers ride `body_b64`; the Rack body arrives
3501
3808
  # BINARY-tagged (see `RuntimeShared.utf8_text`).
3502
- text = RuntimeShared.utf8_text(is_text ? decode_response_bom(raw) : raw)
3809
+ bom_charset = nil
3810
+ text =
3811
+ if is_text
3812
+ decoded, bom_charset = decode_response_bom(raw)
3813
+ RuntimeShared.utf8_text(decoded)
3814
+ else
3815
+ RuntimeShared.utf8_text(raw)
3816
+ end
3503
3817
  out = {
3504
3818
  'status' => status,
3505
3819
  'headers' => hdrs,
@@ -3508,29 +3822,66 @@ module Capybara
3508
3822
  'redirected' => redirected,
3509
3823
  'type' => 'basic'
3510
3824
  }
3825
+ # The BOM-detected encoding (if any) — a frame load pins its document's
3826
+ # characterSet to it (see __csimFrameWindow); highest-precedence signal.
3827
+ out['charset'] = bom_charset if bom_charset
3511
3828
  out['body_b64'] = Base64.strict_encode64(raw) unless is_text
3512
3829
  out
3513
3830
  end
3514
3831
 
3515
- # Strip + decode a single leading byte-order mark, mapping the body to a
3516
- # UTF-8 Ruby string. No BOM → return the bytes untouched (the hot path:
3517
- # just a 2–3 byte prefix check). One BOM is consumed; any further BOMs are
3518
- # ordinary U+FEFF characters in the decoded text (per spec the parser does
3519
- # not strip them again).
3832
+ # Strip + decode a single leading byte-order mark, returning
3833
+ # `[utf8_text, charset]` `charset` is the BOM-selected Encoding-standard
3834
+ # name (highest-precedence encoding signal) or nil when there's no BOM (the
3835
+ # hot path: just a 2–3 byte prefix check). One BOM is consumed; any further
3836
+ # BOMs are ordinary U+FEFF characters in the decoded text (per spec the
3837
+ # parser does not strip them again).
3838
+ # An XML-family document (XHTML / SVG / application+text/xml). Its encoding
3839
+ # default is UTF-8 — the windows-1252 locale default is HTML-only.
3840
+ def xml_content_type?(content_type)
3841
+ mime = content_type.to_s.split(';', 2).first.to_s.strip.downcase
3842
+ mime.end_with?('+xml') || mime == 'application/xml' || mime == 'text/xml'
3843
+ end
3844
+
3845
+ # Does the response carry an explicit encoding signal (so the default
3846
+ # windows-1252 decode must NOT apply)? A `charset=` in the Content-Type, or
3847
+ # a `<meta charset>` / `<meta http-equiv=content-type … charset=…>` in the
3848
+ # HTML prescan window (the first 1024 bytes, per the HTML sniffing algorithm).
3849
+ # The `charset` must start a real attribute / content-charset (preceded by
3850
+ # whitespace, a quote, or `;`), so hyphenated look-alikes — `data-charset=`,
3851
+ # `accept-charset=` — don't false-trigger the signal.
3852
+ def html_charset_signal?(content_type, raw)
3853
+ return true if /;\s*charset\s*=/i.match?(content_type.to_s)
3854
+ head = raw.to_s.b[0, 1024].to_s
3855
+ /<meta\b[^>]*[\s"';]charset\s*=/i.match?(head)
3856
+ end
3857
+
3858
+ # Decode bytes as windows-1252 (the HTML locale-default encoding) to a UTF-8
3859
+ # Ruby string. Replaces undefined slots rather than raising.
3860
+ def decode_windows1252(s)
3861
+ s.to_s.b.dup.force_encoding(Encoding::WINDOWS_1252)
3862
+ .encode(Encoding::UTF_8, invalid: :replace, undef: :replace)
3863
+ rescue StandardError
3864
+ RuntimeShared.utf8_text(s)
3865
+ end
3866
+
3520
3867
  def decode_response_bom(s)
3521
3868
  b = s.b
3522
3869
  if b.start_with?("\xEF\xBB\xBF".b)
3523
- b.byteslice(3..).force_encoding(Encoding::UTF_8)
3870
+ [b.byteslice(3..).force_encoding(Encoding::UTF_8), 'UTF-8']
3524
3871
  elsif b.start_with?("\xFF\xFE".b) || b.start_with?("\xFE\xFF".b)
3525
3872
  # Generic UTF-16: the BOM picks endianness and is dropped by the decoder.
3526
3873
  # Replace malformed units rather than raising (a truncated/odd-length
3527
3874
  # body still yields readable UTF-8 instead of falling back to raw bytes).
3528
- b.force_encoding(Encoding::UTF_16).encode(Encoding::UTF_8, invalid: :replace, undef: :replace)
3875
+ # A UTF-32LE BOM (FF FE 00 00) is matched here as UTF-16LE too — which is
3876
+ # exactly what browsers do (UTF-32 unsupported; the leading FF FE is read
3877
+ # as the UTF-16LE BOM).
3878
+ charset = b.start_with?("\xFF\xFE".b) ? 'UTF-16LE' : 'UTF-16BE'
3879
+ [b.force_encoding(Encoding::UTF_16).encode(Encoding::UTF_8, invalid: :replace, undef: :replace), charset]
3529
3880
  else
3530
- s
3881
+ [s, nil]
3531
3882
  end
3532
3883
  rescue StandardError
3533
- s
3884
+ [s, nil]
3534
3885
  end
3535
3886
 
3536
3887
  def text_response?(headers)
@@ -3566,10 +3917,52 @@ module Capybara
3566
3917
  # the page, discarding all JS state.
3567
3918
  if pure_fragment_navigation?(url)
3568
3919
  update_current_hash(url)
3920
+ elsif @current_realm_id
3921
+ # A JS-driven `location.*` from inside a `within_frame` block
3922
+ # navigates the FRAME, not the top page (same as a self-targeted
3923
+ # link/form there). Gated on the realm, so the main-page path is
3924
+ # untouched.
3925
+ navigate_frame(url)
3569
3926
  else
3570
3927
  navigate(url)
3571
3928
  end
3572
3929
  end
3930
+ # A nested browsing context navigating its OWN `location` (the frame's
3931
+ # `location.href`/assign/replace/`location=`, incl. cross-frame
3932
+ # `iframe.contentWindow.location.href = …`). `realm_id` is the frame's realm.
3933
+ # Deferred like location_assign: applying it re-navigates the owning iframe,
3934
+ # which disposes that realm — illegal while the frame's location setter is
3935
+ # still on the V8 stack — so we stash and drain from `tick_real_time`.
3936
+ def frame_navigate_self(url, realm_id)
3937
+ return if realm_id.nil? || realm_id.zero?
3938
+ # Keyed by realm id (last URL wins per frame) so two different frames each
3939
+ # navigating in one turn both apply — a single slot would drop one.
3940
+ (@pending_frame_nav ||= {})[realm_id] = url.to_s
3941
+ end
3942
+ def consume_pending_frame_nav
3943
+ return if @pending_frame_nav.nil? || @pending_frame_nav.empty?
3944
+ navs = @pending_frame_nav
3945
+ @pending_frame_nav = nil
3946
+ navs.each do |realm_id, url|
3947
+ invalidate_find_cache
3948
+ # If this frame is on the entered `within_frame` stack, navigate it
3949
+ # through `navigate_frame` — it does the full fetch (redirects /
3950
+ # downloads / cookies) AND updates `@frame_stack` / `@current_realm_id`
3951
+ # so the enclosing `within_frame` block sees the new document. Otherwise
3952
+ # (a parent's `iframe.contentWindow.location.href = …`) re-navigate the
3953
+ # owning iframe by realm id via the src-reassignment path. Top-level
3954
+ # frames live in the main document; a nested non-entered frame's element
3955
+ # is in its parent realm's DOM (not yet routed — documented gap).
3956
+ entry = @frame_stack.find {|e| e[:realm_id] == realm_id }
3957
+ if entry
3958
+ navigate_frame(url, entry: entry)
3959
+ else
3960
+ @runtime.call('__csimNavigateFrameByRealm', realm_id, url)
3961
+ end
3962
+ rescue StandardError => e
3963
+ log_console('warn', "frame self-navigation failed: #{e.message}")
3964
+ end
3965
+ end
3573
3966
  # Mirror of `location_assign`'s deferral for `location.reload()`:
3574
3967
  # the JS call lands here from `__locationReload`; running
3575
3968
  # `browser.refresh` directly would `navigate` (rebuilding the
@@ -3582,10 +3975,156 @@ module Capybara
3582
3975
  @pending_reload = false
3583
3976
  refresh
3584
3977
  end
3978
+ # `frame.contentWindow.location.reload()` from a nested browsing context.
3979
+ # Like `frame_navigate_self`, the JS side flags the initiating realm here
3980
+ # and we defer (so the child realm isn't disposed mid-reload()). Keyed by
3981
+ # realm id so two frames reloading in one turn both apply.
3982
+ def frame_reload_self(realm_id)
3983
+ return if realm_id.nil? || realm_id.zero?
3984
+ (@pending_frame_reload ||= []) << realm_id
3985
+ end
3986
+ def consume_pending_frame_reload
3987
+ return if @pending_frame_reload.nil? || @pending_frame_reload.empty?
3988
+ realm_ids = @pending_frame_reload.uniq
3989
+ @pending_frame_reload = nil
3990
+ realm_ids.each do |realm_id|
3991
+ invalidate_find_cache
3992
+ # An entered `within_frame` frame reloads through `navigate_frame` (keeps
3993
+ # the frame stack in sync) — re-fetching its current document URL, which
3994
+ # we read from the still-alive realm. (A blob: URL entered this way is
3995
+ # re-fetched through Rack and so does NOT reuse retained bytes — reloading
3996
+ # an *entered* revoked-blob frame is an accepted gap; the common parent-
3997
+ # held path below reuses bytes via reloadFrame.) Otherwise (a parent's
3998
+ # `iframe.contentWindow.location.reload()`, empty href, or a realm torn
3999
+ # down between flag and drain) re-navigate the owning iframe by realm id
4000
+ # JS-side, reusing the retained content so blob bytes survive a revoke.
4001
+ entry = @frame_stack.find {|e| e[:realm_id] == realm_id }
4002
+ url = entry && @runtime.frame_realm_alive?(realm_id) ? @runtime.realm_call(realm_id, '__csimLocationHref').to_s : ''
4003
+ if entry && !url.empty?
4004
+ navigate_frame(url, entry: entry)
4005
+ else
4006
+ @runtime.call('__csimReloadFrameByRealm', realm_id)
4007
+ end
4008
+ rescue StandardError => e
4009
+ log_console('warn', "frame self-reload failed: #{e.message}")
4010
+ end
4011
+ end
4012
+ # A <form> submitted from INSIDE a nested browsing context (a frame realm
4013
+ # reached via `contentWindow`, not an entered `within_frame` block). The
4014
+ # pending-submit slot lives on the initiating realm's globalThis, which no
4015
+ # top-page drain reads, so the JS side flags the realm here (mirrors
4016
+ # `frame_navigate_self`). Keyed by realm id; deferred + drained from
4017
+ # `drain_pending_navigation` so we never serialize/navigate while the
4018
+ # form's `submit()` is still on the V8 stack.
4019
+ def frame_submit_self(realm_id)
4020
+ return if realm_id.nil? || realm_id.zero?
4021
+ (@pending_frame_submit ||= []) << realm_id
4022
+ end
4023
+ def consume_pending_frame_submit
4024
+ return if @pending_frame_submit.nil? || @pending_frame_submit.empty?
4025
+ realm_ids = @pending_frame_submit.uniq
4026
+ @pending_frame_submit = nil
4027
+ realm_ids.each do |realm_id|
4028
+ next unless @runtime.frame_realm_alive?(realm_id)
4029
+ sub = @runtime.realm_call(realm_id, '__csimTakePendingFormSubmit')
4030
+ next unless sub.is_a?(Hash) && sub['formHandle']
4031
+ invalidate_find_cache
4032
+ submit_form_in_realm(realm_id, sub['formHandle'], sub['submitterHandle'])
4033
+ rescue StandardError => e
4034
+ log_console('warn', "nested-context form submission failed: #{e.message}")
4035
+ end
4036
+ end
4037
+ # Serialize + route a form submitted inside frame realm `realm_id`. We
4038
+ # serialize in the INITIATING realm (so shadow-tree controls are excluded
4039
+ # and relative URLs resolve against that document), then route by target:
4040
+ # - a NAMED frame within that context (a sibling iframe) — reassign its src;
4041
+ # - self / _self / '' — navigate the initiating frame itself, same as a
4042
+ # self-targeted link there (within_frame → navigate_frame; a frame
4043
+ # reached via contentWindow → re-navigate its owning iframe by realm id).
4044
+ # GET fully supported. POST to a self frame needs the entered stack
4045
+ # (navigate_frame_post); POST-to-named and other targets from a nested
4046
+ # context aren't modeled (no in-scope need) — logged rather than dropped.
4047
+ def submit_form_in_realm(realm_id, form_handle, submitter_handle)
4048
+ spec = @runtime.realm_call(realm_id, '__csimFormSerialize', form_handle, submitter_handle || 0)
4049
+ return unless spec.is_a?(Hash)
4050
+ method = spec['method'].to_s.upcase
4051
+ method = 'GET' if method.empty?
4052
+ target = spec['target'].to_s
4053
+ action = spec['action'].to_s
4054
+ fields = (spec['fields'] || []).map {|pair| [pair[0].to_s, pair[1].to_s] }
4055
+ # Non-multipart file inputs contribute the filename only (mirror submit_form_handle's GET path).
4056
+ (spec['fileInputs'] || []).each do |fi|
4057
+ picks = @file_picks && @file_picks[fi['handle'].to_i] || []
4058
+ fields << [fi['name'].to_s, picks.first ? File.basename(picks.first) : '']
4059
+ end
4060
+ body = URI.encode_www_form(fields)
4061
+ get_url = form_get_url(action, body)
4062
+ if frame_self_target?(target)
4063
+ navigate_realm_self(realm_id, get_url, action, method, body, spec['enctype'].to_s)
4064
+ elsif %w[_parent _top _blank].include?(target.downcase)
4065
+ log_console('warn', "nested-context form submit (target=#{target.inspect}) is not modeled")
4066
+ elsif method == 'GET'
4067
+ # Named sibling frame, GET. realm_call returns false when no frame of
4068
+ # that name exists in the initiating document (e.g. it lives in an
4069
+ # ancestor/top context, which HTML target resolution would reach but
4070
+ # we don't); surface it rather than dropping silently.
4071
+ found = @runtime.realm_call(realm_id, '__csimNavigateNamedFrame', target, get_url)
4072
+ log_console('warn', "nested-context form submit: no frame named #{target.inspect} in the submitting document") unless found
4073
+ else
4074
+ log_console('warn', "nested-context form submit (target=#{target.inspect}, method=POST) is not modeled")
4075
+ end
4076
+ end
4077
+ # HTML form-submission "mutate action URL" for GET: REPLACE the action
4078
+ # URL's query with the serialized entry list (dropping any pre-existing
4079
+ # query), preserving a trailing #fragment. String-based so it works on the
4080
+ # raw (possibly relative) action attribute without URI.parse fragility;
4081
+ # the absolute equivalent of submit_form_handle's `uri.query = body`.
4082
+ def form_get_url(action, body)
4083
+ return action if body.empty?
4084
+ base, _hash, frag = action.partition('#')
4085
+ path = base.split('?', 2).first
4086
+ url = "#{path}?#{body}"
4087
+ frag.empty? ? url : "#{url}##{frag}"
4088
+ end
4089
+ # Navigate the initiating frame realm itself (a self-targeted form submit).
4090
+ def navigate_realm_self(realm_id, get_url, action, method, body, enctype)
4091
+ entry = @frame_stack.find {|e| e[:realm_id] == realm_id }
4092
+ if method == 'GET'
4093
+ if entry
4094
+ navigate_frame(resolve_against_current(get_url), entry: entry)
4095
+ else
4096
+ # A frame reached via contentWindow (not on the entered stack): its
4097
+ # owning iframe lives in the parent document — re-navigate by realm id
4098
+ # (relative get_url resolves against the frame's base on rebuild).
4099
+ @runtime.call('__csimNavigateFrameByRealm', realm_id, get_url)
4100
+ end
4101
+ elsif entry
4102
+ navigate_frame_post(resolve_against_current(action), body, enctype, entry: entry)
4103
+ else
4104
+ log_console('warn', "nested-context self-form POST (realm #{realm_id}) is not modeled")
4105
+ end
4106
+ end
3585
4107
  def drain_pending_navigation
3586
4108
  consume_pending_location
4109
+ consume_pending_frame_nav
4110
+ consume_pending_frame_submit
4111
+ consume_pending_frame_reload
3587
4112
  consume_pending_reload
3588
4113
  consume_pending_history_traverse
4114
+ consume_pending_aux_window
4115
+ end
4116
+
4117
+ # A script-driven `anchor.click()` / `target=_blank` navigation with no
4118
+ # Capybara action behind it (e.g. a WPT test) — open the aux window from the
4119
+ # event-loop drain. Safe mid-call (builds a separate Browser). Same-window /
4120
+ # frame navs are left untouched (handled by drain_after_user_action).
4121
+ def consume_pending_aux_window
4122
+ pending = @runtime.call('__csimTakePendingAuxWindow')
4123
+ return unless pending.is_a?(Hash) && pending['url'] && @driver.respond_to?(:open_aux_window)
4124
+ @driver.open_aux_window(resolve_against_current(pending['url'].to_s, use_base: true),
4125
+ source: self, blob_snapshot: pending['blob'])
4126
+ rescue StandardError => e
4127
+ log_console('warn', "aux-window open failed: #{e.message}")
3589
4128
  end
3590
4129
  # POST-after-POST resubmits with the original body; GET-after-GET
3591
4130
  # is just a re-GET. Replay the current history entry.
@@ -3724,11 +4263,11 @@ module Capybara
3724
4263
  # Mirrors `navigate` / `navigate_post`'s fetch + redirect-follow but
3725
4264
  # terminates in `reload_current_frame_realm` instead of a main-page boot.
3726
4265
 
3727
- def navigate_frame(url, depth: 0)
4266
+ def navigate_frame(url, depth: 0, entry: @frame_stack.last)
3728
4267
  raise 'too many redirects' if depth > 10
3729
4268
  invalidate_find_cache
3730
4269
  if url.to_s.match?(%r{\Aabout:blank(?:[?#]|\z)}i)
3731
- reload_current_frame_realm('about:blank', '', 'text/html')
4270
+ reload_current_frame_realm('about:blank', '', 'text/html', entry: entry)
3732
4271
  return
3733
4272
  end
3734
4273
  env = Rack::MockRequest.env_for(url, method: 'GET')
@@ -3738,16 +4277,16 @@ module Capybara
3738
4277
  if (loc = redirect_location(status, headers))
3739
4278
  next_url = carry_fragment(url, resolve_against_current(loc))
3740
4279
  body.close if body.respond_to?(:close)
3741
- return navigate_frame(next_url, depth: depth + 1)
4280
+ return navigate_frame(next_url, depth: depth + 1, entry: entry)
3742
4281
  end
3743
4282
  if download_response?(headers)
3744
4283
  save_downloaded_response(url, headers, body)
3745
4284
  return
3746
4285
  end
3747
- reload_current_frame_realm(url.to_s, read_rack_body(body), response_content_type(headers))
4286
+ reload_current_frame_realm(url.to_s, read_rack_body(body), response_content_type(headers), entry: entry)
3748
4287
  end
3749
4288
 
3750
- def navigate_frame_post(url, body, content_type, depth: 0)
4289
+ def navigate_frame_post(url, body, content_type, depth: 0, entry: @frame_stack.last)
3751
4290
  raise 'too many redirects' if depth > 10
3752
4291
  invalidate_find_cache
3753
4292
  env = Rack::MockRequest.env_for(url, method: 'POST', input: body)
@@ -3761,31 +4300,53 @@ module Capybara
3761
4300
  resp_body.close if resp_body.respond_to?(:close)
3762
4301
  # 301/302/303 → GET; 307/308 preserve method + body (same as navigate_post).
3763
4302
  if [307, 308].include?(status)
3764
- return navigate_frame_post(next_url, body, content_type, depth: depth + 1)
4303
+ return navigate_frame_post(next_url, body, content_type, depth: depth + 1, entry: entry)
3765
4304
  else
3766
- return navigate_frame(next_url, depth: depth + 1)
4305
+ return navigate_frame(next_url, depth: depth + 1, entry: entry)
3767
4306
  end
3768
4307
  end
3769
4308
  if download_response?(headers)
3770
4309
  save_downloaded_response(url, headers, resp_body)
3771
4310
  return
3772
4311
  end
3773
- reload_current_frame_realm(url.to_s, read_rack_body(resp_body), response_content_type(headers))
3774
- end
3775
-
3776
- # Tear down the active frame's realm and rebuild it from `html`, then
3777
- # re-point the iframe element at the new realm. The iframe lives in the
3778
- # PARENT realm, so the rebind host fn runs there.
3779
- def reload_current_frame_realm(url, html, content_type)
3780
- entry = @frame_stack.last
4312
+ reload_current_frame_realm(url.to_s, read_rack_body(resp_body), response_content_type(headers), entry: entry)
4313
+ end
4314
+
4315
+ # Tear down a frame's realm and rebuild it from `html`, then re-point the
4316
+ # iframe element at the new realm. The iframe lives in the PARENT realm, so
4317
+ # the rebind host fn runs there. `entry` defaults to the active frame; a
4318
+ # `_parent`-targeted navigation passes an ancestor entry instead — every
4319
+ # frame below it in the stack is destroyed along with the ancestor's old
4320
+ # document, so we dispose those realms and leave `@current_realm_id` on the
4321
+ # (now-gone) current frame, surfacing StaleElement for the rest of the open
4322
+ # `within_frame` block. Its `ensure` pops back to `entry`, whose `realm_id`
4323
+ # we've updated to the rebuilt realm.
4324
+ #
4325
+ # Teardown reaches the realms on the entered `@frame_stack` (the ones a
4326
+ # find could route into). Like the self-nav path, descendant realms of the
4327
+ # rebuilt frame that were entered-then-popped earlier (so they no longer
4328
+ # sit on the stack) aren't disposed here — they linger, unreferenced and
4329
+ # un-stepped, until the next full-page rebuild's `dispose_frame_realms`. A
4330
+ # bounded per-test leak, only reachable by re-entering a sibling subframe
4331
+ # before an ancestor `_parent` nav; not worth a JS descendant walk on this
4332
+ # path's perf budget.
4333
+ def reload_current_frame_realm(url, html, content_type, entry: @frame_stack.last)
3781
4334
  return unless entry
3782
4335
  old_id = entry[:realm_id]
3783
4336
  parent = entry[:parent_realm_id]
3784
4337
  new_id = @runtime.reload_frame_realm(old_id, parent.to_i, url, RuntimeShared.utf8_text(html), content_type).to_i
3785
4338
  return if new_id.zero?
3786
4339
  rebind_frame_realm(parent, entry[:iframe_handle], old_id, new_id)
3787
- entry[:realm_id] = new_id
3788
- @current_realm_id = new_id
4340
+ if entry.equal?(@frame_stack.last)
4341
+ entry[:realm_id] = new_id
4342
+ @current_realm_id = new_id
4343
+ else
4344
+ # Match by object identity (the branch was chosen by `equal?`); index
4345
+ # by `==` could collide if two entries were ever structurally equal.
4346
+ idx = @frame_stack.index {|e| e.equal?(entry) }
4347
+ @frame_stack[(idx + 1)..].each {|descendant| @runtime.dispose_frame_realm(descendant[:realm_id]) }
4348
+ entry[:realm_id] = new_id
4349
+ end
3789
4350
  invalidate_find_cache
3790
4351
  settle
3791
4352
  end
@@ -3912,11 +4473,31 @@ module Capybara
3912
4473
  # `within_frame` scope is now stale — fall back to the main document.
3913
4474
  reset_frame_scope
3914
4475
  reset_timer_state
3915
- # The DOCUMENT is text; the Rack body arrives BINARY-tagged (see
3916
- # `RuntimeShared.utf8_text`). Charset-header-driven decode is the
3917
- # fuller story; UTF-8 + scrub matches observable browser behavior
3918
- # for the suites we run.
3919
- html = RuntimeShared.utf8_text(html)
4476
+ # The response content type drives both the parser choice (XML vs HTML —
4477
+ # XHTML/XML/SVG parse case-sensitively, no html/head/body skeleton,
4478
+ # `isHtmlDocument` false) and the encoding's HTTP-charset signal.
4479
+ ct = (@last_response_headers || {}).find {|k, _| k.to_s.downcase == 'content-type' }&.last
4480
+ ct = ct.first if ct.is_a?(Array)
4481
+ # HTML document encoding sniffing (the body arrives BINARY-tagged; see
4482
+ # `RuntimeShared.utf8_text`). A leading BOM wins (over <meta charset>) and
4483
+ # is stripped. Otherwise, for an HTML document with NO encoding signal — no
4484
+ # charset in the Content-Type AND no <meta charset> in the prescan — the
4485
+ # locale default is windows-1252 and the bytes decode as such; there is NO
4486
+ # UTF-8 sniffing (WPT encoding/sniffing). A declared charset keeps the
4487
+ # UTF-8 + scrub path (the JS side reports it from the meta; a declared
4488
+ # non-UTF-8 multibyte charset is still UTF-8-decoded — legacy multibyte
4489
+ # tables are out of scope). The windows-1252 default is HTML-only: an XML
4490
+ # document (XHTML/SVG/application+text/xml) defaults to UTF-8, and an empty
4491
+ # body (about:blank, a blank 200) stays UTF-8 too.
4492
+ decoded, doc_charset = decode_response_bom(html)
4493
+ if doc_charset
4494
+ html = RuntimeShared.utf8_text(decoded)
4495
+ elsif html.to_s.empty? || xml_content_type?(ct) || html_charset_signal?(ct, html)
4496
+ html = RuntimeShared.utf8_text(html)
4497
+ else
4498
+ html = decode_windows1252(html)
4499
+ doc_charset = 'windows-1252'
4500
+ end
3920
4501
  opts = {
3921
4502
  'traceActive' => !@trace.nil?,
3922
4503
  'timezone' => ENV['TZ'].to_s,
@@ -3924,12 +4505,13 @@ module Capybara
3924
4505
  'url' => @current_url.to_s,
3925
4506
  'html' => html
3926
4507
  }
3927
- # Carry the response content type so the JS side can pick the XML vs
3928
- # HTML parser (XHTML / XML / SVG documents parse case-sensitively, with
3929
- # no html/head/body skeleton, and report `isHtmlDocument` false).
3930
- ct = (@last_response_headers || {}).find {|k, _| k.to_s.downcase == 'content-type' }&.last
3931
- ct = ct.first if ct.is_a?(Array)
3932
4508
  opts['contentType'] = ct.to_s if ct && !ct.to_s.empty?
4509
+ # The detected document encoding pins document.characterSet (over meta).
4510
+ opts['charset'] = doc_charset if doc_charset
4511
+ # `document.lastModified` reflects the response Last-Modified header (parsed
4512
+ # to local time); absent → the current time (handled JS-side).
4513
+ lm = response_headers['Last-Modified'] # response_headers normalizes keys to Capitalized-Dash form
4514
+ opts['lastModified'] = lm if lm && !lm.to_s.empty?
3933
4515
  if @viewport_width && @viewport_height
3934
4516
  opts['viewportW'] = @viewport_width
3935
4517
  opts['viewportH'] = @viewport_height
@@ -3950,9 +4532,12 @@ module Capybara
3950
4532
  # timer can't abort loading the next page (the page it would affect is
3951
4533
  # being discarded on the very next line).
3952
4534
  def flush_outgoing_page_init
3953
- saved_location = @pending_location
3954
- saved_reload = @pending_reload
3955
- saved_traverse = @pending_history_traverse
4535
+ saved_location = @pending_location
4536
+ saved_reload = @pending_reload
4537
+ saved_traverse = @pending_history_traverse
4538
+ saved_frame_nav = @pending_frame_nav
4539
+ saved_frame_submit = @pending_frame_submit
4540
+ saved_frame_reload = @pending_frame_reload
3956
4541
  begin
3957
4542
  @runtime.run_loop_step(0, SETTLE_MAX_ITER_TASKS, yield_on_gen: false)
3958
4543
  rescue StandardError
@@ -3961,6 +4546,13 @@ module Capybara
3961
4546
  @pending_location = saved_location
3962
4547
  @pending_reload = saved_reload
3963
4548
  @pending_history_traverse = saved_traverse
4549
+ # Don't let a frame-nav / frame-submit / frame-reload intent stashed by
4550
+ # an outgoing-page timer leak into the fresh page — its realm id belongs
4551
+ # to the discarded page (and a reused context id could mis-fire against
4552
+ # an unrelated realm on the new page).
4553
+ @pending_frame_nav = saved_frame_nav
4554
+ @pending_frame_submit = saved_frame_submit
4555
+ @pending_frame_reload = saved_frame_reload
3964
4556
  end
3965
4557
  end
3966
4558