@ultimat3/scraping 19.2.0 → 19.3.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +5 -5
- package/src/error-throws.ts +34 -2
- package/src/errors.ts +7 -0
- package/src/hosts.ts +15 -1
- package/src/http-redirect.ts +78 -0
- package/src/http.ts +152 -38
- package/src/index.ts +3 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@ultimat3/scraping",
|
|
3
|
-
"version": "19.2
|
|
3
|
+
"version": "19.3.2",
|
|
4
4
|
"description": "Browser automation as a job: scrape() returns a JobHandle",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"type": "module",
|
|
@@ -30,9 +30,9 @@
|
|
|
30
30
|
"test": "bun test"
|
|
31
31
|
},
|
|
32
32
|
"dependencies": {
|
|
33
|
-
"@ultimat3/core": "19.2
|
|
34
|
-
"@ultimat3/jobs": "19.2
|
|
35
|
-
"@ultimat3/schema": "19.2
|
|
36
|
-
"@ultimat3/storage": "19.2
|
|
33
|
+
"@ultimat3/core": "19.3.2",
|
|
34
|
+
"@ultimat3/jobs": "19.3.2",
|
|
35
|
+
"@ultimat3/schema": "19.3.2",
|
|
36
|
+
"@ultimat3/storage": "19.3.2"
|
|
37
37
|
}
|
|
38
38
|
}
|
package/src/error-throws.ts
CHANGED
|
@@ -3,7 +3,13 @@
|
|
|
3
3
|
// a browser is rendered through core's `renderThrowable`, which is the rule `bun run error-render`
|
|
4
4
|
// enforces.
|
|
5
5
|
|
|
6
|
-
import {
|
|
6
|
+
import {
|
|
7
|
+
isFixShellSafe,
|
|
8
|
+
isRetryableStatus,
|
|
9
|
+
renderFixShellArg,
|
|
10
|
+
renderThrowable,
|
|
11
|
+
UltimateError,
|
|
12
|
+
} from '@ultimat3/core';
|
|
7
13
|
import type { CaptureClip } from './capture-clip';
|
|
8
14
|
import { ScrapeError } from './errors';
|
|
9
15
|
|
|
@@ -99,7 +105,18 @@ export const profileLocked = (profileDir: string): ScrapeError =>
|
|
|
99
105
|
new ScrapeError({
|
|
100
106
|
code: 'X_SCRAPE_PROFILE_LOCKED',
|
|
101
107
|
cause: `another browser process holds the profile at ${profileDir}`,
|
|
102
|
-
|
|
108
|
+
// `profileDir` is the app's — `localBrowser({ profileDir })`, and a multi-tenant run composes
|
|
109
|
+
// it from a tenant id — and this is the one line in this package that leads with `rm`.
|
|
110
|
+
//
|
|
111
|
+
// A COMMAND or PROSE, never a mixture of the two, and both halves were wrong. The trailing
|
|
112
|
+
// sentence was bare text after an `rm` operand, so pasting the line deleted `once`, `no`,
|
|
113
|
+
// `browser` and four more names out of the current directory; and for a path a shell would
|
|
114
|
+
// read, the placeholder that replaced it — `<the profile directory …>` — is redirection
|
|
115
|
+
// syntax rather than a path, so the "command" could not run at all. So the safe branch puts
|
|
116
|
+
// the explanation behind a `#` and the unsafe one drops the command entirely.
|
|
117
|
+
fix: isFixShellSafe(profileDir)
|
|
118
|
+
? `rm -f ${renderFixShellArg(profileDir, '')}/SingletonLock # only once no browser is using it, or give this run its own localBrowser({ profileDir })`
|
|
119
|
+
: 'delete the SingletonLock file inside the profile directory the cause names, once no browser is using it — or give this run its own localBrowser({ profileDir })',
|
|
103
120
|
meta: { profileDir },
|
|
104
121
|
});
|
|
105
122
|
|
|
@@ -320,6 +337,21 @@ export const bodyTooLarge = (url: string, readBytes: number, maxBytes: number):
|
|
|
320
337
|
meta: { url, readBytes, maxBytes },
|
|
321
338
|
});
|
|
322
339
|
|
|
340
|
+
/**
|
|
341
|
+
* The chain never settled. Its own code rather than `X_SCRAPE_HTTP_FAILED`, because the two send
|
|
342
|
+
* their reader to different places: that one means the site answered and said no, this one means
|
|
343
|
+
* the site kept pointing somewhere else and the leg stopped asking. `hops` is quoted so the cause
|
|
344
|
+
* is falsifiable against the ring — `page.network()` holds one entry per hop, each with the URL
|
|
345
|
+
* that was really requested rather than the one the caller wrote.
|
|
346
|
+
*/
|
|
347
|
+
export const redirectLoop = (url: string, lastUrl: string, hops: number): ScrapeError =>
|
|
348
|
+
new ScrapeError({
|
|
349
|
+
code: 'X_SCRAPE_REDIRECT_LOOP',
|
|
350
|
+
cause: `${url} redirected ${String(hops)} times without answering; the last hop pointed at ${lastUrl}`,
|
|
351
|
+
fix: 'request the URL the chain settles on — page.network() lists every hop this leg took — or repair the session: a chain that never ends is a login redirecting to a page that redirects back to the login',
|
|
352
|
+
meta: { url, lastUrl, hops },
|
|
353
|
+
});
|
|
354
|
+
|
|
323
355
|
/**
|
|
324
356
|
* TERMINAL, and the retry table cannot be talked out of it. A site that locks an account after
|
|
325
357
|
* three wrong attempts turns a retrying framework into the thing that destroys the user's
|
package/src/errors.ts
CHANGED
|
@@ -33,6 +33,7 @@ export const SCRAPE_OWNED_ERROR_CODES = [
|
|
|
33
33
|
'X_SCRAPE_CAPTURE_INVALID',
|
|
34
34
|
'X_SCRAPE_HTTP_FAILED',
|
|
35
35
|
'X_SCRAPE_BODY_TOO_LARGE',
|
|
36
|
+
'X_SCRAPE_REDIRECT_LOOP',
|
|
36
37
|
'X_SCRAPE_AUTH_FAILED',
|
|
37
38
|
'X_SCRAPE_SESSION_EXPIRED',
|
|
38
39
|
'X_SCRAPE_PROMPT_UNANSWERED',
|
|
@@ -83,6 +84,7 @@ export const SCRAPE_ERROR_TITLES: Readonly<Record<ScrapeOwnedErrorCode, string>>
|
|
|
83
84
|
X_SCRAPE_CAPTURE_INVALID: 'the capture names a framing no picture can be taken with',
|
|
84
85
|
X_SCRAPE_HTTP_FAILED: 'the site answered the HTTP leg with a non-2xx status',
|
|
85
86
|
X_SCRAPE_BODY_TOO_LARGE: 'the HTTP response body passed its byte cap',
|
|
87
|
+
X_SCRAPE_REDIRECT_LOOP: 'the HTTP leg followed its hop limit of redirects without an answer',
|
|
86
88
|
X_SCRAPE_AUTH_FAILED: 'the credentials were rejected',
|
|
87
89
|
X_SCRAPE_SESSION_EXPIRED: 'the restored session is no longer valid and nothing can renew it',
|
|
88
90
|
X_SCRAPE_PROMPT_UNANSWERED: 'a login step asked for a code and nothing answered',
|
|
@@ -150,6 +152,11 @@ export const SCRAPE_ERROR_RETRY = {
|
|
|
150
152
|
// A response size is a property of the endpoint, not of the moment: attempt 2 buffers the same
|
|
151
153
|
// gigabyte and dies the same way. The fix is a number on the request, so a human decides it.
|
|
152
154
|
X_SCRAPE_BODY_TOO_LARGE: 'terminal',
|
|
155
|
+
// A chain that does not settle is a property of the endpoint and of the session that reaches it,
|
|
156
|
+
// not of the moment: attempt 2 restores the same cookies and walks the same ten hops. The
|
|
157
|
+
// reachable case is a login redirecting to a page that redirects back to the login, and the
|
|
158
|
+
// repair is a credential or a URL — a human's edit, not a queue's second try.
|
|
159
|
+
X_SCRAPE_REDIRECT_LOOP: 'terminal',
|
|
153
160
|
X_SCRAPE_YIELD_COLLAPSED: 'terminal',
|
|
154
161
|
// A declaration error, raised by `scrape()` before any attempt exists — there is no run to
|
|
155
162
|
// retry, and the same definition would refuse identically forever.
|
package/src/hosts.ts
CHANGED
|
@@ -12,7 +12,20 @@ export const ANY_HOST: HostRule = '*';
|
|
|
12
12
|
* Schemes with no host to match. `about:blank` is where every browser starts, `data:` and `blob:`
|
|
13
13
|
* never leave the process — refusing them would refuse the first page load of every run.
|
|
14
14
|
*/
|
|
15
|
-
const HOSTLESS_SCHEMES = new Set(['about:', 'data:', 'blob:'
|
|
15
|
+
const HOSTLESS_SCHEMES = new Set(['about:', 'data:', 'blob:']);
|
|
16
|
+
|
|
17
|
+
/**
|
|
18
|
+
* Hostless AND refused, which is why it is not on the line above — it sat there until 2026-09.
|
|
19
|
+
* `javascript:` has no host for the same reason `data:` has none, and that is the whole
|
|
20
|
+
* resemblance: the other three are inert content this process renders, while this one is code
|
|
21
|
+
* EXECUTED in the current document's origin, with the session's cookies and the session's
|
|
22
|
+
* `localStorage` already in scope. `allowHosts` cannot say anything about a URL with no host to
|
|
23
|
+
* name, so "no host, therefore allowed" was the allow list opting itself out of the one navigation
|
|
24
|
+
* that needs no host to exfiltrate through — `javascript:fetch('/admin').then(post_elsewhere)` is
|
|
25
|
+
* a same-origin read on an allow-listed site. Fail closed; there is no legitimate scrape verb that
|
|
26
|
+
* needs it (`page.eval` is the declared seam).
|
|
27
|
+
*/
|
|
28
|
+
const REFUSED_SCHEMES = new Set(['javascript:']);
|
|
16
29
|
|
|
17
30
|
export interface HostDecision {
|
|
18
31
|
readonly allowed: boolean;
|
|
@@ -44,6 +57,7 @@ export function hostMatches(host: string, rule: HostRule): boolean {
|
|
|
44
57
|
*/
|
|
45
58
|
export function hostDecision(url: string, allowHosts: readonly HostRule[]): HostDecision {
|
|
46
59
|
const scheme = url.slice(0, Math.max(0, url.indexOf(':') + 1)).toLowerCase();
|
|
60
|
+
if (REFUSED_SCHEMES.has(scheme)) return { allowed: false, host: '' };
|
|
47
61
|
if (HOSTLESS_SCHEMES.has(scheme)) return { allowed: true, host: '' };
|
|
48
62
|
let host: string;
|
|
49
63
|
try {
|
|
@@ -0,0 +1,78 @@
|
|
|
1
|
+
// The redirect chain as a decision this leg makes hop by hop, rather than one `fetch` follows on
|
|
2
|
+
// its own. A followed redirect is a request to a URL nobody screened: `allowHosts`, robots and the
|
|
3
|
+
// rate limit are all asked about the url the caller wrote, and the answer comes back from wherever
|
|
4
|
+
// the site pointed. This file is the arithmetic — which status is a hop, where it goes, and what
|
|
5
|
+
// method carries — so `http.ts` re-applies the same three gates per hop and nothing else changes.
|
|
6
|
+
|
|
7
|
+
/**
|
|
8
|
+
* The five statuses that carry a `Location` a client may follow. `300` is deliberately absent:
|
|
9
|
+
* multiple choices has no single target, and picking one for the caller is guessing.
|
|
10
|
+
*/
|
|
11
|
+
const REDIRECT_STATUSES = new Set([301, 302, 303, 307, 308]);
|
|
12
|
+
|
|
13
|
+
/**
|
|
14
|
+
* The hop ceiling, and a NAMED one so the refusal can quote it. Ten is the number every major
|
|
15
|
+
* client settled on (curl's `--max-redirs`, the fetch spec's own counter) — high enough that no
|
|
16
|
+
* honest CDN chain reaches it, low enough that a login bouncing to a page that bounces back costs
|
|
17
|
+
* ten requests instead of a worker.
|
|
18
|
+
*/
|
|
19
|
+
export const MAX_REDIRECT_HOPS = 10;
|
|
20
|
+
|
|
21
|
+
/** What the next hop is asked with: a redirect can change the method and drop the body. */
|
|
22
|
+
export interface RedirectHop {
|
|
23
|
+
readonly url: string;
|
|
24
|
+
readonly method: string;
|
|
25
|
+
readonly body: string | undefined;
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/**
|
|
29
|
+
* `303` means "look over there with a GET", and `301`/`302` after a POST mean the same thing in
|
|
30
|
+
* every browser and every HTTP client shipped since — the spec permits the rewrite and universal
|
|
31
|
+
* practice performs it. `307`/`308` exist precisely to say "re-send exactly what you sent", so
|
|
32
|
+
* they keep both. Getting this wrong is not cosmetic: re-POSTing an order body at the URL a `303`
|
|
33
|
+
* points to is a second order.
|
|
34
|
+
*
|
|
35
|
+
* The rewrite is per METHOD as well as per status, which "anything that is not a GET or a HEAD
|
|
36
|
+
* loses its body" got wrong in the other direction: the fetch standard rewrites `301`/`302` for
|
|
37
|
+
* POST alone, so a `PUT` or a `DELETE` re-asked as a bodyless GET is the caller's write silently
|
|
38
|
+
* not happening — one read, a 200, and nothing changed at the target.
|
|
39
|
+
*/
|
|
40
|
+
const carriesBody = (status: number, method: string): boolean => {
|
|
41
|
+
if (status === 307 || status === 308) return true;
|
|
42
|
+
// Case-folded because `fetch` normalises a standard method on the way out — `post` leaves as
|
|
43
|
+
// `POST` — so a decision keyed on the caller's spelling would rewrite one and keep the other.
|
|
44
|
+
const verb = method.toUpperCase();
|
|
45
|
+
// `303` is the status that rewrites every method BUT those two, and it spares `HEAD` on purpose:
|
|
46
|
+
// a HEAD promoted to a GET fetches the body the caller said it did not want.
|
|
47
|
+
if (status === 303) return verb === 'GET' || verb === 'HEAD';
|
|
48
|
+
return verb !== 'POST';
|
|
49
|
+
};
|
|
50
|
+
|
|
51
|
+
/**
|
|
52
|
+
* The next hop, or `undefined` when this response IS the answer. A `Location` that will not parse
|
|
53
|
+
* against the hop it came from is `undefined` too, and that is the fail-closed side: a URL this
|
|
54
|
+
* package cannot resolve is a URL it cannot screen, so it is never requested — the 3xx becomes the
|
|
55
|
+
* returned response and `X_SCRAPE_HTTP_FAILED` reports it with its own status.
|
|
56
|
+
*/
|
|
57
|
+
export function redirectHop(
|
|
58
|
+
status: number,
|
|
59
|
+
location: string | null,
|
|
60
|
+
from: string,
|
|
61
|
+
method: string,
|
|
62
|
+
body: string | undefined,
|
|
63
|
+
): RedirectHop | undefined {
|
|
64
|
+
if (!REDIRECT_STATUSES.has(status)) return undefined;
|
|
65
|
+
if (location === null || location.trim() === '') return undefined;
|
|
66
|
+
let resolved: string;
|
|
67
|
+
try {
|
|
68
|
+
resolved = new URL(location, from).toString();
|
|
69
|
+
} catch {
|
|
70
|
+
return undefined;
|
|
71
|
+
}
|
|
72
|
+
const keep = carriesBody(status, method);
|
|
73
|
+
return {
|
|
74
|
+
url: resolved,
|
|
75
|
+
method: keep ? method : 'GET',
|
|
76
|
+
body: keep ? body : undefined,
|
|
77
|
+
};
|
|
78
|
+
}
|
package/src/http.ts
CHANGED
|
@@ -15,7 +15,9 @@ import type { StandardSchemaV1 } from '@ultimat3/schema';
|
|
|
15
15
|
import { parse } from '@ultimat3/schema';
|
|
16
16
|
import type { ScrapeClock } from './clock';
|
|
17
17
|
import { cookieHeaderFor } from './cookie-scope';
|
|
18
|
-
import { bodyTooLarge, hostBlocked, httpFailed, scrapeTimeout } from './error-throws';
|
|
18
|
+
import { bodyTooLarge, hostBlocked, httpFailed, redirectLoop, scrapeTimeout } from './error-throws';
|
|
19
|
+
import type { RedirectHop } from './http-redirect';
|
|
20
|
+
import { MAX_REDIRECT_HOPS, redirectHop } from './http-redirect';
|
|
19
21
|
import type { InterceptRules } from './intercept';
|
|
20
22
|
import { interceptVerdict } from './intercept';
|
|
21
23
|
import type { NetworkRing } from './rings';
|
|
@@ -163,13 +165,93 @@ export function responseOver(
|
|
|
163
165
|
};
|
|
164
166
|
}
|
|
165
167
|
|
|
168
|
+
/**
|
|
169
|
+
* A hop's body is thrown away — nothing reads a redirect's — and an unread stream holds its socket
|
|
170
|
+
* open until the collector gets to it. `cancel()` on a body that already errored rejects, and that
|
|
171
|
+
* rejection is not this request's failure: the hop is over and the next one is what the caller is
|
|
172
|
+
* waiting for.
|
|
173
|
+
*/
|
|
174
|
+
const discardHopBody = async (response: Response): Promise<void> => {
|
|
175
|
+
try {
|
|
176
|
+
await response.body?.cancel();
|
|
177
|
+
} catch {
|
|
178
|
+
// Discarded bytes cannot fail a request.
|
|
179
|
+
}
|
|
180
|
+
};
|
|
181
|
+
|
|
182
|
+
/**
|
|
183
|
+
* The credentials a CALLER set, which are scoped to the origin they were set for — the platform's
|
|
184
|
+
* own `follow` deleted these on a cross-origin hop (step 13 of the fetch standard's HTTP-redirect
|
|
185
|
+
* fetch) and this file took the chain over, so this file owns the strip. `cookie` is here for the
|
|
186
|
+
* hand-written header only: the jar's own value is computed per hop by `cookieHeaderFor`, which
|
|
187
|
+
* has always been scoped to the host being dialled.
|
|
188
|
+
*/
|
|
189
|
+
const CROSS_ORIGIN_STRIPPED = new Set(['authorization', 'proxy-authorization', 'cookie']);
|
|
190
|
+
|
|
191
|
+
const withoutCredentials = (headers: Readonly<Record<string, string>>): Record<string, string> =>
|
|
192
|
+
Object.fromEntries(
|
|
193
|
+
Object.entries(headers).filter(([name]) => !CROSS_ORIGIN_STRIPPED.has(name.toLowerCase())),
|
|
194
|
+
);
|
|
195
|
+
|
|
196
|
+
/** Unparseable is not same-origin: a URL this package cannot read is one it cannot vouch for. */
|
|
197
|
+
const sameOrigin = (left: string, right: string): boolean => {
|
|
198
|
+
try {
|
|
199
|
+
return new URL(left).origin === new URL(right).origin;
|
|
200
|
+
} catch {
|
|
201
|
+
return false;
|
|
202
|
+
}
|
|
203
|
+
};
|
|
204
|
+
|
|
205
|
+
/**
|
|
206
|
+
* The final answer, read under the cap. Counted as it arrives rather than `.text()`, which
|
|
207
|
+
* materialises first and checks never: a 30s stream at 50MB/s is a 1.5GB allocation the worker
|
|
208
|
+
* does not get back, and it takes every other job on that worker with it. The same read
|
|
209
|
+
* `robots-fetch.ts` performs.
|
|
210
|
+
*/
|
|
211
|
+
const readResponse = async (
|
|
212
|
+
url: string,
|
|
213
|
+
response: Response,
|
|
214
|
+
maxBytes: number,
|
|
215
|
+
secrets: ScrapeSecrets | undefined,
|
|
216
|
+
): Promise<ScrapeResponse> => {
|
|
217
|
+
const capped = await readWithinLimit(response.body, maxBytes);
|
|
218
|
+
if ('over' in capped) throw bodyTooLarge(url, capped.over, maxBytes);
|
|
219
|
+
const body = new TextDecoder().decode(capped.bytes);
|
|
220
|
+
return responseOver(
|
|
221
|
+
url,
|
|
222
|
+
response.status,
|
|
223
|
+
headerRecord(response.headers),
|
|
224
|
+
() => Promise.resolve(body),
|
|
225
|
+
secrets,
|
|
226
|
+
);
|
|
227
|
+
};
|
|
228
|
+
|
|
166
229
|
/**
|
|
167
230
|
* The real transport. Every guarantee the page makes is re-applied here, in the same order and
|
|
168
231
|
* through the same functions — `interceptVerdict` is the one host rule, `RobotsGate` is the one
|
|
169
232
|
* robots rule, and neither is re-implemented for the second leg.
|
|
233
|
+
*
|
|
234
|
+
* REDIRECTS ARE FOLLOWED BY THIS FILE, hop by hop, and not by `fetch`. Until 2026-09 the call
|
|
235
|
+
* carried the platform default `redirect: 'follow'`, so an allow-listed endpoint answering
|
|
236
|
+
* `302 -> http://169.254.169.254/…` read the metadata service THROUGH the allow list: the three
|
|
237
|
+
* gates above had all been asked about the URL the caller wrote, `res.url` and the network ring
|
|
238
|
+
* both reported that URL, and nothing ever asked robots about where the body came from. The CDP
|
|
239
|
+
* leg never had the hole — interception fires per hop there — and this file's header claims parity
|
|
240
|
+
* with it.
|
|
170
241
|
*/
|
|
171
242
|
export function httpOverFetch(init: HttpTransportInit): ScrapeHttp {
|
|
172
243
|
const call: ScrapeFetch = init.fetch ?? fetch;
|
|
244
|
+
/**
|
|
245
|
+
* The three gates, in one place, so the initial URL and hop seven are screened by the same code
|
|
246
|
+
* in the same order. A second copy for redirects is how the two drift.
|
|
247
|
+
*/
|
|
248
|
+
const screen = async (target: string): Promise<void> => {
|
|
249
|
+
if (interceptVerdict(target, 'fetch', init.rules) !== 'allow') {
|
|
250
|
+
throw hostBlocked(target, init.rules.allowHosts);
|
|
251
|
+
}
|
|
252
|
+
await init.robots?.assertAllowed(target);
|
|
253
|
+
await init.pace?.(init.signal);
|
|
254
|
+
};
|
|
173
255
|
return {
|
|
174
256
|
async request(url: string, request: HttpRequestInit = {}): Promise<ScrapeResponse> {
|
|
175
257
|
// Screened FIRST — before the activity touch, before the robots read this method performs
|
|
@@ -196,52 +278,84 @@ export function httpOverFetch(init: HttpTransportInit): ScrapeHttp {
|
|
|
196
278
|
1,
|
|
197
279
|
);
|
|
198
280
|
init.onActivity?.();
|
|
199
|
-
|
|
200
|
-
throw hostBlocked(url, init.rules.allowHosts);
|
|
201
|
-
}
|
|
202
|
-
await init.robots?.assertAllowed(url);
|
|
203
|
-
await init.pace?.(init.signal);
|
|
281
|
+
await screen(url);
|
|
204
282
|
const session = await init.session();
|
|
205
|
-
const cookies = cookieHeaderFor(session.cookies, url);
|
|
206
283
|
// `AbortSignal.timeout` and NOT `clock.sleep`: this is a deadline handed to the platform's
|
|
207
284
|
// own fetch, not a wait this package performs — and under a test clock a slept deadline
|
|
208
285
|
// would fire on the microtask after it was armed, cancelling every request instantly.
|
|
209
286
|
// The offline transport (`http-recorded.ts`) is what a test runs, and it has no deadline.
|
|
287
|
+
//
|
|
288
|
+
// ONE deadline for the whole chain, not one per hop: the caller declared a budget for
|
|
289
|
+
// getting an answer, and ten hops each allowed the full budget is ten times the wait.
|
|
210
290
|
const deadlineSignal = AbortSignal.timeout(timeoutMs);
|
|
211
291
|
const signals = init.signal === undefined ? [deadlineSignal] : [deadlineSignal, init.signal];
|
|
212
292
|
try {
|
|
213
|
-
|
|
214
|
-
method: request.method ?? 'GET',
|
|
215
|
-
headers: {
|
|
216
|
-
...session.headers,
|
|
217
|
-
...(session.userAgent === '' ? {} : { 'user-agent': session.userAgent }),
|
|
218
|
-
...(cookies === undefined ? {} : { cookie: cookies }),
|
|
219
|
-
...request.headers,
|
|
220
|
-
},
|
|
221
|
-
...(request.body === undefined ? {} : { body: request.body }),
|
|
222
|
-
signal: AbortSignal.any(signals),
|
|
223
|
-
...(init.proxy === undefined ? {} : { proxy: init.proxy }),
|
|
224
|
-
});
|
|
225
|
-
init.network.push({
|
|
226
|
-
method: request.method ?? 'GET',
|
|
227
|
-
url,
|
|
228
|
-
status: response.status,
|
|
229
|
-
resourceType: 'fetch',
|
|
230
|
-
at: init.clock.now().getTime(),
|
|
231
|
-
});
|
|
232
|
-
// Counted as it arrives rather than `.text()`, which materialises first and checks never:
|
|
233
|
-
// a 30s stream at 50MB/s is a 1.5GB allocation the worker does not get back, and it takes
|
|
234
|
-
// every other job on that worker with it. The same read `robots-fetch.ts` performs.
|
|
235
|
-
const capped = await readWithinLimit(response.body, maxBytes);
|
|
236
|
-
if ('over' in capped) throw bodyTooLarge(url, capped.over, maxBytes);
|
|
237
|
-
const body = new TextDecoder().decode(capped.bytes);
|
|
238
|
-
return responseOver(
|
|
293
|
+
let hop: RedirectHop = {
|
|
239
294
|
url,
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
295
|
+
method: request.method ?? 'GET',
|
|
296
|
+
body: request.body,
|
|
297
|
+
};
|
|
298
|
+
// Latched off at the first cross-origin hop and never back on: the fetch standard DELETES
|
|
299
|
+
// the header from the request rather than re-deciding per hop, so `A -> B -> A` does not
|
|
300
|
+
// hand A's bearer back on the way home.
|
|
301
|
+
let credentialsInScope = true;
|
|
302
|
+
for (let followed = 0; ; followed += 1) {
|
|
303
|
+
// Per hop, because the jar is every domain the browser touched: a cookie computed for
|
|
304
|
+
// the first host and re-sent to the second is the leak `cookie-scope.ts` exists to
|
|
305
|
+
// prevent, arriving through the one door that never asked it twice.
|
|
306
|
+
const cookies = cookieHeaderFor(session.cookies, hop.url);
|
|
307
|
+
// Both header sources re-scoped the way the jar above already is, in the SAME precedence
|
|
308
|
+
// they had before: an `authorization` minted for the first host was the one credential
|
|
309
|
+
// that still rode along to wherever a `302` pointed.
|
|
310
|
+
const carried = credentialsInScope
|
|
311
|
+
? session.headers
|
|
312
|
+
: withoutCredentials(session.headers);
|
|
313
|
+
const declared = credentialsInScope
|
|
314
|
+
? request.headers
|
|
315
|
+
: withoutCredentials(request.headers ?? {});
|
|
316
|
+
const response = await call(hop.url, {
|
|
317
|
+
method: hop.method,
|
|
318
|
+
headers: {
|
|
319
|
+
...carried,
|
|
320
|
+
...(session.userAgent === '' ? {} : { 'user-agent': session.userAgent }),
|
|
321
|
+
...(cookies === undefined ? {} : { cookie: cookies }),
|
|
322
|
+
...declared,
|
|
323
|
+
},
|
|
324
|
+
...(hop.body === undefined ? {} : { body: hop.body }),
|
|
325
|
+
signal: AbortSignal.any(signals),
|
|
326
|
+
// The whole point: the platform's own `follow` is what made the four lines above
|
|
327
|
+
// decorative, because it dials the target itself and hands back one Response.
|
|
328
|
+
redirect: 'manual',
|
|
329
|
+
...(init.proxy === undefined ? {} : { proxy: init.proxy }),
|
|
330
|
+
});
|
|
331
|
+
// The URL that was REQUESTED, hop by hop. A chain reported under the caller's URL sends
|
|
332
|
+
// its reader hunting for a request the site never answered.
|
|
333
|
+
init.network.push({
|
|
334
|
+
method: hop.method,
|
|
335
|
+
url: hop.url,
|
|
336
|
+
status: response.status,
|
|
337
|
+
resourceType: 'fetch',
|
|
338
|
+
at: init.clock.now().getTime(),
|
|
339
|
+
});
|
|
340
|
+
const next = redirectHop(
|
|
341
|
+
response.status,
|
|
342
|
+
response.headers.get('location'),
|
|
343
|
+
hop.url,
|
|
344
|
+
hop.method,
|
|
345
|
+
hop.body,
|
|
346
|
+
);
|
|
347
|
+
if (next === undefined)
|
|
348
|
+
return await readResponse(hop.url, response, maxBytes, init.secrets);
|
|
349
|
+
// Discarded BEFORE the refusal, not after it: a throw over an unread stream holds that
|
|
350
|
+
// hop's socket until the collector reaches it, and the refusal path is exactly the one a
|
|
351
|
+
// hostile chain drives ten times per request.
|
|
352
|
+
await discardHopBody(response);
|
|
353
|
+
if (followed >= MAX_REDIRECT_HOPS) throw redirectLoop(url, next.url, followed + 1);
|
|
354
|
+
init.onActivity?.();
|
|
355
|
+
await screen(next.url);
|
|
356
|
+
if (!sameOrigin(hop.url, next.url)) credentialsInScope = false;
|
|
357
|
+
hop = next;
|
|
358
|
+
}
|
|
245
359
|
} catch (thrown) {
|
|
246
360
|
// A deadline that fired is this package's own timeout, with its own code and fix — never
|
|
247
361
|
// the platform's bare `TimeoutError` reaching a job's retry classifier unclassified.
|
package/src/index.ts
CHANGED
|
@@ -68,6 +68,7 @@ export {
|
|
|
68
68
|
profileLocked,
|
|
69
69
|
promptUnanswered,
|
|
70
70
|
recoverRefused,
|
|
71
|
+
redirectLoop,
|
|
71
72
|
remoteRequired,
|
|
72
73
|
robotsDisallowed,
|
|
73
74
|
scrapeNotImplemented,
|
|
@@ -112,6 +113,8 @@ export type { HttpRequestInit, HttpTransportInit, ScrapeHttp, ScrapeResponse } f
|
|
|
112
113
|
export { DEFAULT_HTTP_MAX_BYTES, httpOverFetch, responseOver } from './http';
|
|
113
114
|
export type { HttpRecordingLookup, RecordedHttpInit } from './http-recorded';
|
|
114
115
|
export { httpRecordingFilename, httpRecordingsOf, recordedHttp } from './http-recorded';
|
|
116
|
+
export type { RedirectHop } from './http-redirect';
|
|
117
|
+
export { MAX_REDIRECT_HOPS, redirectHop } from './http-redirect';
|
|
115
118
|
export type { InterceptRules, InterceptVerdict } from './intercept';
|
|
116
119
|
export { interceptVerdict, refusalEntry } from './intercept';
|
|
117
120
|
export type { OfflineSessionInit } from './offline-session';
|