@ultimat3/scraping 19.1.3 → 19.3.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@ultimat3/scraping",
3
- "version": "19.1.3",
3
+ "version": "19.3.1",
4
4
  "description": "Browser automation as a job: scrape() returns a JobHandle",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -30,9 +30,9 @@
30
30
  "test": "bun test"
31
31
  },
32
32
  "dependencies": {
33
- "@ultimat3/core": "19.1.3",
34
- "@ultimat3/jobs": "19.1.3",
35
- "@ultimat3/schema": "19.1.3",
36
- "@ultimat3/storage": "19.1.3"
33
+ "@ultimat3/core": "19.3.1",
34
+ "@ultimat3/jobs": "19.3.1",
35
+ "@ultimat3/schema": "19.3.1",
36
+ "@ultimat3/storage": "19.3.1"
37
37
  }
38
38
  }
@@ -3,7 +3,13 @@
3
3
  // a browser is rendered through core's `renderThrowable`, which is the rule `bun run error-render`
4
4
  // enforces.
5
5
 
6
- import { isRetryableStatus, renderThrowable, UltimateError } from '@ultimat3/core';
6
+ import {
7
+ isFixShellSafe,
8
+ isRetryableStatus,
9
+ renderFixShellArg,
10
+ renderThrowable,
11
+ UltimateError,
12
+ } from '@ultimat3/core';
7
13
  import type { CaptureClip } from './capture-clip';
8
14
  import { ScrapeError } from './errors';
9
15
 
@@ -36,19 +42,81 @@ export const cdpAttachFailed = (cdpUrl: string, thrown: unknown): ScrapeError =>
36
42
  meta: { cdpUrl },
37
43
  });
38
44
 
45
+ /**
46
+ * What separates "there is no browser" from "the browser stopped answering". Both arrive at the
47
+ * same `catch` around `launch()` (`driver-cdp.ts`), and only one of them is worth a second attempt.
48
+ *
49
+ * Matched on the LAUNCHER'S OWN WORDS rather than on a flag this function could be passed, because
50
+ * the launch site hands it one `unknown` and nothing else. Every entry below is a message a browser
51
+ * library or the kernel emits for a browser that is not there, and none of them is producible by a
52
+ * page that merely went away mid-run — which is what keeps the split honest at `cdp-target.ts`'s
53
+ * call site, where the browser demonstrably existed.
54
+ *
55
+ * Lowercase, and compared against a lowercased render, so a library re-capitalising its own message
56
+ * does not silently turn a permanent misconfiguration back into five retries.
57
+ */
58
+ const BROWSER_ABSENT_SIGNS: readonly string[] = [
59
+ // puppeteer-core ships no browser and has no default: this is its whole message when nothing was
60
+ // configured — ``An `executablePath` or `channel` must be specified for `puppeteer-core```.
61
+ 'must be specified',
62
+ // `spawn /usr/bin/google-chrome ENOENT` — an executablePath naming a file that is not there,
63
+ // which is what a CHROME_PATH left over from another machine is.
64
+ 'enoent',
65
+ // The binary is there and this uid may not execute it. Attempt 2 runs as the same user.
66
+ 'eacces',
67
+ // The two spellings puppeteer has shipped for "I looked for a browser and found none".
68
+ 'could not find chrome',
69
+ 'could not find browser',
70
+ 'was not found at the configured executablepath',
71
+ ];
72
+
73
+ const browserIsAbsent = (thrown: unknown): boolean => {
74
+ const text = renderThrowable(thrown).toLowerCase();
75
+ return BROWSER_ABSENT_SIGNS.some((sign) => text.includes(sign));
76
+ };
77
+
78
+ /**
79
+ * A launch or a page call that failed. TWO codes, because the two failures have opposite answers.
80
+ *
81
+ * `X_SCRAPE_BROWSER_UNREACHABLE` is transport and stays retryable: a browser that was there is
82
+ * gone, and attempt 2 may find it back.
83
+ *
84
+ * `X_SCRAPE_BROWSER_MISSING` is a misconfiguration and is terminal. Its fix used to read "raise
85
+ * watchdog: { idleMs } on the scrape() definition … otherwise re-run once the browser host is
86
+ * back", which on the `x shot` path names a definition that does not exist and an act that never
87
+ * helps — while the retry table spent five browser launches proving it.
88
+ */
39
89
  export const browserUnreachable = (driver: string, thrown: unknown): ScrapeError =>
40
- new ScrapeError({
41
- code: 'X_SCRAPE_BROWSER_UNREACHABLE',
42
- cause: `the ${driver} browser stopped answering: ${renderThrowable(thrown)}`,
43
- fix: 'raise watchdog: { idleMs } on the scrape() definition if the site is genuinely slow, otherwise re-run once the browser host is back',
44
- meta: { driver },
45
- });
90
+ browserIsAbsent(thrown)
91
+ ? new ScrapeError({
92
+ code: 'X_SCRAPE_BROWSER_MISSING',
93
+ cause: `the ${driver} driver has no browser to launch: ${renderThrowable(thrown)}`,
94
+ fix: 'install a Chrome or Chromium and name it — localBrowser({ launcher, executablePath: env.CHROME_PATH }) — or attach to one somebody else is running with remoteBrowser({ cdpUrl: env.SCRAPE_CDP_URL })',
95
+ meta: { driver },
96
+ })
97
+ : new ScrapeError({
98
+ code: 'X_SCRAPE_BROWSER_UNREACHABLE',
99
+ cause: `the ${driver} browser stopped answering: ${renderThrowable(thrown)}`,
100
+ fix: 'raise watchdog: { idleMs } on the scrape() definition if the site is genuinely slow, otherwise re-run once the browser host is back',
101
+ meta: { driver },
102
+ });
46
103
 
47
104
  export const profileLocked = (profileDir: string): ScrapeError =>
48
105
  new ScrapeError({
49
106
  code: 'X_SCRAPE_PROFILE_LOCKED',
50
107
  cause: `another browser process holds the profile at ${profileDir}`,
51
- fix: `rm -f ${profileDir}/SingletonLock once no browser is using it, or give this run its own localBrowser({ profileDir })`,
108
+ // `profileDir` is the app's — `localBrowser({ profileDir })`, and a multi-tenant run composes
109
+ // it from a tenant id — and this is the one line in this package that leads with `rm`.
110
+ //
111
+ // A COMMAND or PROSE, never a mixture of the two, and both halves were wrong. The trailing
112
+ // sentence was bare text after an `rm` operand, so pasting the line deleted `once`, `no`,
113
+ // `browser` and four more names out of the current directory; and for a path a shell would
114
+ // read, the placeholder that replaced it — `<the profile directory …>` — is redirection
115
+ // syntax rather than a path, so the "command" could not run at all. So the safe branch puts
116
+ // the explanation behind a `#` and the unsafe one drops the command entirely.
117
+ fix: isFixShellSafe(profileDir)
118
+ ? `rm -f ${renderFixShellArg(profileDir, '')}/SingletonLock # only once no browser is using it, or give this run its own localBrowser({ profileDir })`
119
+ : 'delete the SingletonLock file inside the profile directory the cause names, once no browser is using it — or give this run its own localBrowser({ profileDir })',
52
120
  meta: { profileDir },
53
121
  });
54
122
 
@@ -269,6 +337,21 @@ export const bodyTooLarge = (url: string, readBytes: number, maxBytes: number):
269
337
  meta: { url, readBytes, maxBytes },
270
338
  });
271
339
 
340
+ /**
341
+ * The chain never settled. Its own code rather than `X_SCRAPE_HTTP_FAILED`, because the two send
342
+ * their reader to different places: that one means the site answered and said no, this one means
343
+ * the site kept pointing somewhere else and the leg stopped asking. `hops` is quoted so the cause
344
+ * is falsifiable against the ring — `page.network()` holds one entry per hop, each with the URL
345
+ * that was really requested rather than the one the caller wrote.
346
+ */
347
+ export const redirectLoop = (url: string, lastUrl: string, hops: number): ScrapeError =>
348
+ new ScrapeError({
349
+ code: 'X_SCRAPE_REDIRECT_LOOP',
350
+ cause: `${url} redirected ${String(hops)} times without answering; the last hop pointed at ${lastUrl}`,
351
+ fix: 'request the URL the chain settles on — page.network() lists every hop this leg took — or repair the session: a chain that never ends is a login redirecting to a page that redirects back to the login',
352
+ meta: { url, lastUrl, hops },
353
+ });
354
+
272
355
  /**
273
356
  * TERMINAL, and the retry table cannot be talked out of it. A site that locks an account after
274
357
  * three wrong attempts turns a retrying framework into the thing that destroys the user's
package/src/errors.ts CHANGED
@@ -11,6 +11,7 @@ export const SCRAPE_OWNED_ERROR_CODES = [
11
11
  'X_SCRAPE_DRIVER_UNKNOWN',
12
12
  'X_SCRAPE_CDP_ATTACH_FAILED',
13
13
  'X_SCRAPE_BROWSER_UNREACHABLE',
14
+ 'X_SCRAPE_BROWSER_MISSING',
14
15
  'X_SCRAPE_PROFILE_LOCKED',
15
16
  'X_SCRAPE_HOST_BLOCKED',
16
17
  'X_SCRAPE_SELECTOR_MISSING',
@@ -32,6 +33,7 @@ export const SCRAPE_OWNED_ERROR_CODES = [
32
33
  'X_SCRAPE_CAPTURE_INVALID',
33
34
  'X_SCRAPE_HTTP_FAILED',
34
35
  'X_SCRAPE_BODY_TOO_LARGE',
36
+ 'X_SCRAPE_REDIRECT_LOOP',
35
37
  'X_SCRAPE_AUTH_FAILED',
36
38
  'X_SCRAPE_SESSION_EXPIRED',
37
39
  'X_SCRAPE_PROMPT_UNANSWERED',
@@ -59,6 +61,7 @@ export const SCRAPE_ERROR_TITLES: Readonly<Record<ScrapeOwnedErrorCode, string>>
59
61
  X_SCRAPE_DRIVER_UNKNOWN: 'no browser driver is installed for this run',
60
62
  X_SCRAPE_CDP_ATTACH_FAILED: 'the CDP endpoint refused the attach',
61
63
  X_SCRAPE_BROWSER_UNREACHABLE: 'the browser went away mid-run',
64
+ X_SCRAPE_BROWSER_MISSING: 'there is no browser on this machine for the driver to launch',
62
65
  X_SCRAPE_PROFILE_LOCKED: 'another process holds this browser profile',
63
66
  X_SCRAPE_HOST_BLOCKED: 'the page asked for a host allowHosts does not list',
64
67
  X_SCRAPE_SELECTOR_MISSING: 'the selector never appeared inside its window',
@@ -81,6 +84,7 @@ export const SCRAPE_ERROR_TITLES: Readonly<Record<ScrapeOwnedErrorCode, string>>
81
84
  X_SCRAPE_CAPTURE_INVALID: 'the capture names a framing no picture can be taken with',
82
85
  X_SCRAPE_HTTP_FAILED: 'the site answered the HTTP leg with a non-2xx status',
83
86
  X_SCRAPE_BODY_TOO_LARGE: 'the HTTP response body passed its byte cap',
87
+ X_SCRAPE_REDIRECT_LOOP: 'the HTTP leg followed its hop limit of redirects without an answer',
84
88
  X_SCRAPE_AUTH_FAILED: 'the credentials were rejected',
85
89
  X_SCRAPE_SESSION_EXPIRED: 'the restored session is no longer valid and nothing can renew it',
86
90
  X_SCRAPE_PROMPT_UNANSWERED: 'a login step asked for a code and nothing answered',
@@ -106,6 +110,10 @@ registerErrorCodes(
106
110
  */
107
111
  export const SCRAPE_ERROR_RETRY = {
108
112
  X_SCRAPE_CDP_ATTACH_FAILED: 'retryable',
113
+ // Retryable, and it now means ONLY what its title says: a browser that was there and stopped
114
+ // answering. It used to be raised for a launch that failed because there was no browser to
115
+ // launch at all, which the same run repeated five times — `X_SCRAPE_BROWSER_MISSING` below is
116
+ // that half, split out and terminal.
109
117
  X_SCRAPE_BROWSER_UNREACHABLE: 'retryable',
110
118
  X_SCRAPE_TIMEOUT: 'retryable',
111
119
  X_SCRAPE_WEDGED: 'retryable',
@@ -124,6 +132,14 @@ export const SCRAPE_ERROR_RETRY = {
124
132
  // Everything below is terminal, and each one is listed rather than left to the default so that
125
133
  // deleting a line is a visible decision.
126
134
  X_SCRAPE_DRIVER_UNKNOWN: 'terminal',
135
+ // TERMINAL where its sibling `X_SCRAPE_BROWSER_UNREACHABLE` is retryable, and the difference IS
136
+ // the reason the two codes are separate. "The browser host went away" is transport — attempt 2
137
+ // may find it back. "There is no browser binary on this box, or none was configured" is a
138
+ // MISCONFIGURATION: attempt 2 spawns the same absent path, five times, on a five-attempt policy,
139
+ // and on an authenticated target that is five arrivals at a login for no chance of a different
140
+ // answer. Measured on `x shot`, where there is no `scrape()` definition to raise a watchdog on
141
+ // and re-running never helps. The repair is an install or an `executablePath`, so a human decides.
142
+ X_SCRAPE_BROWSER_MISSING: 'terminal',
127
143
  X_SCRAPE_PROFILE_LOCKED: 'terminal',
128
144
  X_SCRAPE_HOST_BLOCKED: 'terminal',
129
145
  X_SCRAPE_SELECTOR_MISSING: 'terminal',
@@ -136,6 +152,11 @@ export const SCRAPE_ERROR_RETRY = {
136
152
  // A response size is a property of the endpoint, not of the moment: attempt 2 buffers the same
137
153
  // gigabyte and dies the same way. The fix is a number on the request, so a human decides it.
138
154
  X_SCRAPE_BODY_TOO_LARGE: 'terminal',
155
+ // A chain that does not settle is a property of the endpoint and of the session that reaches it,
156
+ // not of the moment: attempt 2 restores the same cookies and walks the same ten hops. The
157
+ // reachable case is a login redirecting to a page that redirects back to the login, and the
158
+ // repair is a credential or a URL — a human's edit, not a queue's second try.
159
+ X_SCRAPE_REDIRECT_LOOP: 'terminal',
139
160
  X_SCRAPE_YIELD_COLLAPSED: 'terminal',
140
161
  // A declaration error, raised by `scrape()` before any attempt exists — there is no run to
141
162
  // retry, and the same definition would refuse identically forever.
package/src/hosts.ts CHANGED
@@ -12,7 +12,20 @@ export const ANY_HOST: HostRule = '*';
12
12
  * Schemes with no host to match. `about:blank` is where every browser starts, `data:` and `blob:`
13
13
  * never leave the process — refusing them would refuse the first page load of every run.
14
14
  */
15
- const HOSTLESS_SCHEMES = new Set(['about:', 'data:', 'blob:', 'javascript:']);
15
+ const HOSTLESS_SCHEMES = new Set(['about:', 'data:', 'blob:']);
16
+
17
+ /**
18
+ * Hostless AND refused, which is why it is not on the line above — it sat there until 2026-09.
19
+ * `javascript:` has no host for the same reason `data:` has none, and that is the whole
20
+ * resemblance: the other three are inert content this process renders, while this one is code
21
+ * EXECUTED in the current document's origin, with the session's cookies and the session's
22
+ * `localStorage` already in scope. `allowHosts` cannot say anything about a URL with no host to
23
+ * name, so "no host, therefore allowed" was the allow list opting itself out of the one navigation
24
+ * that needs no host to exfiltrate through — `javascript:fetch('/admin').then(post_elsewhere)` is
25
+ * a same-origin read on an allow-listed site. Fail closed; there is no legitimate scrape verb that
26
+ * needs it (`page.eval` is the declared seam).
27
+ */
28
+ const REFUSED_SCHEMES = new Set(['javascript:']);
16
29
 
17
30
  export interface HostDecision {
18
31
  readonly allowed: boolean;
@@ -44,6 +57,7 @@ export function hostMatches(host: string, rule: HostRule): boolean {
44
57
  */
45
58
  export function hostDecision(url: string, allowHosts: readonly HostRule[]): HostDecision {
46
59
  const scheme = url.slice(0, Math.max(0, url.indexOf(':') + 1)).toLowerCase();
60
+ if (REFUSED_SCHEMES.has(scheme)) return { allowed: false, host: '' };
47
61
  if (HOSTLESS_SCHEMES.has(scheme)) return { allowed: true, host: '' };
48
62
  let host: string;
49
63
  try {
@@ -0,0 +1,78 @@
1
+ // The redirect chain as a decision this leg makes hop by hop, rather than one `fetch` follows on
2
+ // its own. A followed redirect is a request to a URL nobody screened: `allowHosts`, robots and the
3
+ // rate limit are all asked about the url the caller wrote, and the answer comes back from wherever
4
+ // the site pointed. This file is the arithmetic — which status is a hop, where it goes, and what
5
+ // method carries — so `http.ts` re-applies the same three gates per hop and nothing else changes.
6
+
7
+ /**
8
+ * The five statuses that carry a `Location` a client may follow. `300` is deliberately absent:
9
+ * multiple choices has no single target, and picking one for the caller is guessing.
10
+ */
11
+ const REDIRECT_STATUSES = new Set([301, 302, 303, 307, 308]);
12
+
13
+ /**
14
+ * The hop ceiling, and a NAMED one so the refusal can quote it. Ten is the number every major
15
+ * client settled on (curl's `--max-redirs`, the fetch spec's own counter) — high enough that no
16
+ * honest CDN chain reaches it, low enough that a login bouncing to a page that bounces back costs
17
+ * ten requests instead of a worker.
18
+ */
19
+ export const MAX_REDIRECT_HOPS = 10;
20
+
21
+ /** What the next hop is asked with: a redirect can change the method and drop the body. */
22
+ export interface RedirectHop {
23
+ readonly url: string;
24
+ readonly method: string;
25
+ readonly body: string | undefined;
26
+ }
27
+
28
+ /**
29
+ * `303` means "look over there with a GET", and `301`/`302` after a POST mean the same thing in
30
+ * every browser and every HTTP client shipped since — the spec permits the rewrite and universal
31
+ * practice performs it. `307`/`308` exist precisely to say "re-send exactly what you sent", so
32
+ * they keep both. Getting this wrong is not cosmetic: re-POSTing an order body at the URL a `303`
33
+ * points to is a second order.
34
+ *
35
+ * The rewrite is per METHOD as well as per status, which "anything that is not a GET or a HEAD
36
+ * loses its body" got wrong in the other direction: the fetch standard rewrites `301`/`302` for
37
+ * POST alone, so a `PUT` or a `DELETE` re-asked as a bodyless GET is the caller's write silently
38
+ * not happening — one read, a 200, and nothing changed at the target.
39
+ */
40
+ const carriesBody = (status: number, method: string): boolean => {
41
+ if (status === 307 || status === 308) return true;
42
+ // Case-folded because `fetch` normalises a standard method on the way out — `post` leaves as
43
+ // `POST` — so a decision keyed on the caller's spelling would rewrite one and keep the other.
44
+ const verb = method.toUpperCase();
45
+ // `303` is the status that rewrites every method BUT those two, and it spares `HEAD` on purpose:
46
+ // a HEAD promoted to a GET fetches the body the caller said it did not want.
47
+ if (status === 303) return verb === 'GET' || verb === 'HEAD';
48
+ return verb !== 'POST';
49
+ };
50
+
51
+ /**
52
+ * The next hop, or `undefined` when this response IS the answer. A `Location` that will not parse
53
+ * against the hop it came from is `undefined` too, and that is the fail-closed side: a URL this
54
+ * package cannot resolve is a URL it cannot screen, so it is never requested — the 3xx becomes the
55
+ * returned response and `X_SCRAPE_HTTP_FAILED` reports it with its own status.
56
+ */
57
+ export function redirectHop(
58
+ status: number,
59
+ location: string | null,
60
+ from: string,
61
+ method: string,
62
+ body: string | undefined,
63
+ ): RedirectHop | undefined {
64
+ if (!REDIRECT_STATUSES.has(status)) return undefined;
65
+ if (location === null || location.trim() === '') return undefined;
66
+ let resolved: string;
67
+ try {
68
+ resolved = new URL(location, from).toString();
69
+ } catch {
70
+ return undefined;
71
+ }
72
+ const keep = carriesBody(status, method);
73
+ return {
74
+ url: resolved,
75
+ method: keep ? method : 'GET',
76
+ body: keep ? body : undefined,
77
+ };
78
+ }
package/src/http.ts CHANGED
@@ -15,7 +15,9 @@ import type { StandardSchemaV1 } from '@ultimat3/schema';
15
15
  import { parse } from '@ultimat3/schema';
16
16
  import type { ScrapeClock } from './clock';
17
17
  import { cookieHeaderFor } from './cookie-scope';
18
- import { bodyTooLarge, hostBlocked, httpFailed, scrapeTimeout } from './error-throws';
18
+ import { bodyTooLarge, hostBlocked, httpFailed, redirectLoop, scrapeTimeout } from './error-throws';
19
+ import type { RedirectHop } from './http-redirect';
20
+ import { MAX_REDIRECT_HOPS, redirectHop } from './http-redirect';
19
21
  import type { InterceptRules } from './intercept';
20
22
  import { interceptVerdict } from './intercept';
21
23
  import type { NetworkRing } from './rings';
@@ -163,13 +165,93 @@ export function responseOver(
163
165
  };
164
166
  }
165
167
 
168
+ /**
169
+ * A hop's body is thrown away — nothing reads a redirect's — and an unread stream holds its socket
170
+ * open until the collector gets to it. `cancel()` on a body that already errored rejects, and that
171
+ * rejection is not this request's failure: the hop is over and the next one is what the caller is
172
+ * waiting for.
173
+ */
174
+ const discardHopBody = async (response: Response): Promise<void> => {
175
+ try {
176
+ await response.body?.cancel();
177
+ } catch {
178
+ // Discarded bytes cannot fail a request.
179
+ }
180
+ };
181
+
182
+ /**
183
+ * The credentials a CALLER set, which are scoped to the origin they were set for — the platform's
184
+ * own `follow` deleted these on a cross-origin hop (step 13 of the fetch standard's HTTP-redirect
185
+ * fetch) and this file took the chain over, so this file owns the strip. `cookie` is here for the
186
+ * hand-written header only: the jar's own value is computed per hop by `cookieHeaderFor`, which
187
+ * has always been scoped to the host being dialled.
188
+ */
189
+ const CROSS_ORIGIN_STRIPPED = new Set(['authorization', 'proxy-authorization', 'cookie']);
190
+
191
+ const withoutCredentials = (headers: Readonly<Record<string, string>>): Record<string, string> =>
192
+ Object.fromEntries(
193
+ Object.entries(headers).filter(([name]) => !CROSS_ORIGIN_STRIPPED.has(name.toLowerCase())),
194
+ );
195
+
196
+ /** Unparseable is not same-origin: a URL this package cannot read is one it cannot vouch for. */
197
+ const sameOrigin = (left: string, right: string): boolean => {
198
+ try {
199
+ return new URL(left).origin === new URL(right).origin;
200
+ } catch {
201
+ return false;
202
+ }
203
+ };
204
+
205
+ /**
206
+ * The final answer, read under the cap. Counted as it arrives rather than `.text()`, which
207
+ * materialises first and checks never: a 30s stream at 50MB/s is a 1.5GB allocation the worker
208
+ * does not get back, and it takes every other job on that worker with it. The same read
209
+ * `robots-fetch.ts` performs.
210
+ */
211
+ const readResponse = async (
212
+ url: string,
213
+ response: Response,
214
+ maxBytes: number,
215
+ secrets: ScrapeSecrets | undefined,
216
+ ): Promise<ScrapeResponse> => {
217
+ const capped = await readWithinLimit(response.body, maxBytes);
218
+ if ('over' in capped) throw bodyTooLarge(url, capped.over, maxBytes);
219
+ const body = new TextDecoder().decode(capped.bytes);
220
+ return responseOver(
221
+ url,
222
+ response.status,
223
+ headerRecord(response.headers),
224
+ () => Promise.resolve(body),
225
+ secrets,
226
+ );
227
+ };
228
+
166
229
  /**
167
230
  * The real transport. Every guarantee the page makes is re-applied here, in the same order and
168
231
  * through the same functions — `interceptVerdict` is the one host rule, `RobotsGate` is the one
169
232
  * robots rule, and neither is re-implemented for the second leg.
233
+ *
234
+ * REDIRECTS ARE FOLLOWED BY THIS FILE, hop by hop, and not by `fetch`. Until 2026-09 the call
235
+ * carried the platform default `redirect: 'follow'`, so an allow-listed endpoint answering
236
+ * `302 -> http://169.254.169.254/…` read the metadata service THROUGH the allow list: the three
237
+ * gates above had all been asked about the URL the caller wrote, `res.url` and the network ring
238
+ * both reported that URL, and nothing ever asked robots about where the body came from. The CDP
239
+ * leg never had the hole — interception fires per hop there — and this file's header claims parity
240
+ * with it.
170
241
  */
171
242
  export function httpOverFetch(init: HttpTransportInit): ScrapeHttp {
172
243
  const call: ScrapeFetch = init.fetch ?? fetch;
244
+ /**
245
+ * The three gates, in one place, so the initial URL and hop seven are screened by the same code
246
+ * in the same order. A second copy for redirects is how the two drift.
247
+ */
248
+ const screen = async (target: string): Promise<void> => {
249
+ if (interceptVerdict(target, 'fetch', init.rules) !== 'allow') {
250
+ throw hostBlocked(target, init.rules.allowHosts);
251
+ }
252
+ await init.robots?.assertAllowed(target);
253
+ await init.pace?.(init.signal);
254
+ };
173
255
  return {
174
256
  async request(url: string, request: HttpRequestInit = {}): Promise<ScrapeResponse> {
175
257
  // Screened FIRST — before the activity touch, before the robots read this method performs
@@ -196,52 +278,84 @@ export function httpOverFetch(init: HttpTransportInit): ScrapeHttp {
196
278
  1,
197
279
  );
198
280
  init.onActivity?.();
199
- if (interceptVerdict(url, 'fetch', init.rules) !== 'allow') {
200
- throw hostBlocked(url, init.rules.allowHosts);
201
- }
202
- await init.robots?.assertAllowed(url);
203
- await init.pace?.(init.signal);
281
+ await screen(url);
204
282
  const session = await init.session();
205
- const cookies = cookieHeaderFor(session.cookies, url);
206
283
  // `AbortSignal.timeout` and NOT `clock.sleep`: this is a deadline handed to the platform's
207
284
  // own fetch, not a wait this package performs — and under a test clock a slept deadline
208
285
  // would fire on the microtask after it was armed, cancelling every request instantly.
209
286
  // The offline transport (`http-recorded.ts`) is what a test runs, and it has no deadline.
287
+ //
288
+ // ONE deadline for the whole chain, not one per hop: the caller declared a budget for
289
+ // getting an answer, and ten hops each allowed the full budget is ten times the wait.
210
290
  const deadlineSignal = AbortSignal.timeout(timeoutMs);
211
291
  const signals = init.signal === undefined ? [deadlineSignal] : [deadlineSignal, init.signal];
212
292
  try {
213
- const response = await call(url, {
214
- method: request.method ?? 'GET',
215
- headers: {
216
- ...session.headers,
217
- ...(session.userAgent === '' ? {} : { 'user-agent': session.userAgent }),
218
- ...(cookies === undefined ? {} : { cookie: cookies }),
219
- ...request.headers,
220
- },
221
- ...(request.body === undefined ? {} : { body: request.body }),
222
- signal: AbortSignal.any(signals),
223
- ...(init.proxy === undefined ? {} : { proxy: init.proxy }),
224
- });
225
- init.network.push({
226
- method: request.method ?? 'GET',
227
- url,
228
- status: response.status,
229
- resourceType: 'fetch',
230
- at: init.clock.now().getTime(),
231
- });
232
- // Counted as it arrives rather than `.text()`, which materialises first and checks never:
233
- // a 30s stream at 50MB/s is a 1.5GB allocation the worker does not get back, and it takes
234
- // every other job on that worker with it. The same read `robots-fetch.ts` performs.
235
- const capped = await readWithinLimit(response.body, maxBytes);
236
- if ('over' in capped) throw bodyTooLarge(url, capped.over, maxBytes);
237
- const body = new TextDecoder().decode(capped.bytes);
238
- return responseOver(
293
+ let hop: RedirectHop = {
239
294
  url,
240
- response.status,
241
- headerRecord(response.headers),
242
- () => Promise.resolve(body),
243
- init.secrets,
244
- );
295
+ method: request.method ?? 'GET',
296
+ body: request.body,
297
+ };
298
+ // Latched off at the first cross-origin hop and never back on: the fetch standard DELETES
299
+ // the header from the request rather than re-deciding per hop, so `A -> B -> A` does not
300
+ // hand A's bearer back on the way home.
301
+ let credentialsInScope = true;
302
+ for (let followed = 0; ; followed += 1) {
303
+ // Per hop, because the jar is every domain the browser touched: a cookie computed for
304
+ // the first host and re-sent to the second is the leak `cookie-scope.ts` exists to
305
+ // prevent, arriving through the one door that never asked it twice.
306
+ const cookies = cookieHeaderFor(session.cookies, hop.url);
307
+ // Both header sources re-scoped the way the jar above already is, in the SAME precedence
308
+ // they had before: an `authorization` minted for the first host was the one credential
309
+ // that still rode along to wherever a `302` pointed.
310
+ const carried = credentialsInScope
311
+ ? session.headers
312
+ : withoutCredentials(session.headers);
313
+ const declared = credentialsInScope
314
+ ? request.headers
315
+ : withoutCredentials(request.headers ?? {});
316
+ const response = await call(hop.url, {
317
+ method: hop.method,
318
+ headers: {
319
+ ...carried,
320
+ ...(session.userAgent === '' ? {} : { 'user-agent': session.userAgent }),
321
+ ...(cookies === undefined ? {} : { cookie: cookies }),
322
+ ...declared,
323
+ },
324
+ ...(hop.body === undefined ? {} : { body: hop.body }),
325
+ signal: AbortSignal.any(signals),
326
+ // The whole point: the platform's own `follow` is what made the four lines above
327
+ // decorative, because it dials the target itself and hands back one Response.
328
+ redirect: 'manual',
329
+ ...(init.proxy === undefined ? {} : { proxy: init.proxy }),
330
+ });
331
+ // The URL that was REQUESTED, hop by hop. A chain reported under the caller's URL sends
332
+ // its reader hunting for a request the site never answered.
333
+ init.network.push({
334
+ method: hop.method,
335
+ url: hop.url,
336
+ status: response.status,
337
+ resourceType: 'fetch',
338
+ at: init.clock.now().getTime(),
339
+ });
340
+ const next = redirectHop(
341
+ response.status,
342
+ response.headers.get('location'),
343
+ hop.url,
344
+ hop.method,
345
+ hop.body,
346
+ );
347
+ if (next === undefined)
348
+ return await readResponse(hop.url, response, maxBytes, init.secrets);
349
+ // Discarded BEFORE the refusal, not after it: a throw over an unread stream holds that
350
+ // hop's socket until the collector reaches it, and the refusal path is exactly the one a
351
+ // hostile chain drives ten times per request.
352
+ await discardHopBody(response);
353
+ if (followed >= MAX_REDIRECT_HOPS) throw redirectLoop(url, next.url, followed + 1);
354
+ init.onActivity?.();
355
+ await screen(next.url);
356
+ if (!sameOrigin(hop.url, next.url)) credentialsInScope = false;
357
+ hop = next;
358
+ }
245
359
  } catch (thrown) {
246
360
  // A deadline that fired is this package's own timeout, with its own code and fix — never
247
361
  // the platform's bare `TimeoutError` reaching a job's retry classifier unclassified.
package/src/index.ts CHANGED
@@ -68,6 +68,7 @@ export {
68
68
  profileLocked,
69
69
  promptUnanswered,
70
70
  recoverRefused,
71
+ redirectLoop,
71
72
  remoteRequired,
72
73
  robotsDisallowed,
73
74
  scrapeNotImplemented,
@@ -112,6 +113,8 @@ export type { HttpRequestInit, HttpTransportInit, ScrapeHttp, ScrapeResponse } f
112
113
  export { DEFAULT_HTTP_MAX_BYTES, httpOverFetch, responseOver } from './http';
113
114
  export type { HttpRecordingLookup, RecordedHttpInit } from './http-recorded';
114
115
  export { httpRecordingFilename, httpRecordingsOf, recordedHttp } from './http-recorded';
116
+ export type { RedirectHop } from './http-redirect';
117
+ export { MAX_REDIRECT_HOPS, redirectHop } from './http-redirect';
115
118
  export type { InterceptRules, InterceptVerdict } from './intercept';
116
119
  export { interceptVerdict, refusalEntry } from './intercept';
117
120
  export type { OfflineSessionInit } from './offline-session';