@trawlme/cli 3.10.0 → 3.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,167 @@
1
+ /**
2
+ * Pure CDP -> storageState mapping + domain scoping for `scraps account
3
+ * session capture` (trawl_cli#183). No I/O here — every function takes
4
+ * plain data and returns plain data, so the full cookie/localStorage
5
+ * matrix (sameSite casing, expires-in-seconds, domain scoping, …) is
6
+ * testable without a real Chrome.
7
+ *
8
+ * The output shape mirrors trawl_node#1976's ingest validator
9
+ * (modules/scraps/helpers/sessionShape.js) — CookieData fields, `sameSite`
10
+ * in `Strict|Lax|None`, `expires` in seconds, `domain` required. This file
11
+ * pre-normalizes to that shape so the server's validator never has to
12
+ * reject the WHOLE batch over one anomalous cookie the client could have
13
+ * caught itself (see normalizeCdpCookie's sameSite=None handling below).
14
+ */
15
+ /** A cookie as CDP's `Storage.getCookies`/`Network.Cookie` returns it. */
16
+ export interface RawCdpCookie {
17
+ name: string;
18
+ value: string;
19
+ domain: string;
20
+ path?: string;
21
+ /** Unix seconds; -1 denotes a session cookie (matches chrome.cookies shape). */
22
+ expires?: number;
23
+ httpOnly?: boolean;
24
+ secure?: boolean;
25
+ /** CDP's `Network.CookieSameSite` enum — nominally `Strict|Lax|None`, but
26
+ * normalized defensively (case-insensitively) rather than trusted as-is. */
27
+ sameSite?: string;
28
+ [key: string]: unknown;
29
+ }
30
+ /** trawl_node sessionShape.js's `CookieData` — only these fields are ever forwarded. */
31
+ export interface CookieData {
32
+ name: string;
33
+ value: string;
34
+ domain: string;
35
+ path?: string;
36
+ secure?: boolean;
37
+ httpOnly?: boolean;
38
+ sameSite?: 'Strict' | 'Lax' | 'None';
39
+ expires?: number;
40
+ }
41
+ /** One origin's captured localStorage, pre-dedup/pre-scope. */
42
+ export interface RawOriginLocalStorage {
43
+ origin: string;
44
+ entries: Array<{
45
+ name: string;
46
+ value: string;
47
+ }>;
48
+ }
49
+ export interface OriginStorage {
50
+ origin: string;
51
+ localStorage: Array<{
52
+ name: string;
53
+ value: string;
54
+ }>;
55
+ }
56
+ export interface StorageState {
57
+ cookies: CookieData[];
58
+ origins: OriginStorage[];
59
+ }
60
+ /**
61
+ * Domain scoping (trawl_cli#183 review finding 2 — cross-tenant leak). The
62
+ * previous version of this file peeled a hostname down to a "registrable
63
+ * domain" (last 2-3 labels, with a small hardcoded allowlist for
64
+ * `co.uk`-shaped suffixes) and treated any host sharing that suffix as in
65
+ * scope. That is wrong for any multi-tenant hosting domain NOT in the
66
+ * allowlist — `getRegistrableDomain('alice.github.io')` collapsed to
67
+ * `'github.io'`, so `isHostInScope('bob.github.io', 'github.io')` came back
68
+ * `true`: a scrap targeting `alice.github.io` would upload `bob.github.io`'s
69
+ * cookies too (same for `herokuapp.com`, `vercel.app`, `s3.amazonaws.com`,
70
+ * …) — every one of them would need its own allowlist entry, which is
71
+ * exactly the Public Suffix List this file deliberately avoids depending
72
+ * on.
73
+ *
74
+ * The fix NARROWS instead: scope every capture to RFC 6265 §5.1.3
75
+ * domain-matching against the scrap URL's own host (`getTargetHost`) —
76
+ * never a peeled/derived domain, so there is nothing left to get wrong
77
+ * about a given hosting provider's suffix.
78
+ *
79
+ * Verified against a real Chrome 152 (loopback probe, `--host-resolver-rules`
80
+ * remapping `alice.github.io`/`app.example.com`/etc. to 127.0.0.1, no real
81
+ * network traffic):
82
+ * - Chrome REFUSES to let a page at `alice.github.io` set a cookie with
83
+ * `Domain=github.io` at all — it never reaches the cookie jar. That's
84
+ * Chrome's own PSL enforcement on the SET side, and it's what makes the
85
+ * zero-PSL-dependency approach below safe: a public-suffix-scoped cookie
86
+ * simply cannot exist in a real jar to be mis-scoped in the first place.
87
+ * - `Storage.getCookies` DOES distinguish a host-only cookie from an
88
+ * explicit-`Domain=` one at the CDP layer, even when the explicit domain
89
+ * string equals the current host exactly: a bare `document.cookie="k=v"`
90
+ * on `alice.github.io` comes back with `domain:"alice.github.io"` (no
91
+ * leading dot); `document.cookie="k=v; domain=alice.github.io"` — same
92
+ * effective host — comes back `domain:".alice.github.io"` (leading dot).
93
+ * That leading dot is real, load-bearing information: it is exactly
94
+ * RFC 6265's own signal for "this is a domain-match cookie, not a
95
+ * host-only one" and `isCookieDomainInScope` below reads it as such
96
+ * rather than stripping it.
97
+ *
98
+ * One deliberate loss, not a bug: a host-only cookie set on a SIBLING host
99
+ * (e.g. `auth.example.com` while the scrap targets `app.example.com`) is
100
+ * out of scope here. A real browser would never send that cookie to
101
+ * `app.example.com` either — host-only means exactly that host — so
102
+ * nothing replay-relevant is lost; it would only matter if the scrap
103
+ * script itself later navigates to `auth.example.com`, which this capture
104
+ * has no visibility into anyway.
105
+ */
106
+ /**
107
+ * @desc The exact host `targetUrl` will actually be requested at — the
108
+ * single anchor every scope decision in this file is relative to. NOT a
109
+ * peeled "registrable domain": no suffix list, no label-counting, no
110
+ * allowlist to keep up to date.
111
+ */
112
+ export declare function getTargetHost(targetUrl: string): string;
113
+ /**
114
+ * @desc RFC 6265 §5.1.3 domain-matching for one captured cookie's `domain`
115
+ * attribute exactly as CDP's `Storage.getCookies` reports it (leading dot
116
+ * preserved when Chrome ever wrote one, absent for a host-only cookie —
117
+ * see the module doc comment for how that was verified against a real
118
+ * Chrome).
119
+ * - Leading dot (a domain-match cookie): in scope for the target host
120
+ * itself, or any subdomain of it.
121
+ * - No leading dot (a host-only cookie): in scope ONLY for that exact
122
+ * host — never a subdomain, never a sibling, never a parent. This is
123
+ * what excludes `bob.github.io` from a capture targeting
124
+ * `alice.github.io`, and `auth.example.com` from one targeting
125
+ * `app.example.com`.
126
+ */
127
+ export declare function isCookieDomainInScope(rawDomain: string, targetHost: string): boolean;
128
+ /**
129
+ * @desc True when a captured page's own hostname (`originHost` — an
130
+ * origin's localStorage has no "domain attribute" concept, unlike a
131
+ * cookie) is the target host itself, or a subdomain of it. Anchored the
132
+ * opposite way from a host-only cookie check: here the TARGET is the root
133
+ * and the origin must fall under it, so a sibling (`auth.example.com` for
134
+ * a scrap on `app.example.com`) is still excluded, but a genuine
135
+ * subdomain page opened during the same login flow is not.
136
+ */
137
+ export declare function isOriginHostInScope(originHost: string, targetHost: string): boolean;
138
+ export interface MapCookiesResult {
139
+ cookies: CookieData[];
140
+ totalSeen: number;
141
+ droppedOutOfScope: number;
142
+ droppedInvalid: number;
143
+ }
144
+ /**
145
+ * @desc Scope + normalize a raw CDP cookie jar down to the cookies a real
146
+ * browser would actually send to `targetUrl`'s host (`isCookieDomainInScope`
147
+ * — RFC 6265 domain-matching, never a peeled registrable domain). Never
148
+ * throws — a structurally broken cookie is counted and dropped, not fatal
149
+ * to the rest of the capture (see the "batch same class failures" rule:
150
+ * this IS the sweep of the offending class, applied once here rather than
151
+ * duplicated at each call site).
152
+ */
153
+ export declare function mapCookies(rawCookies: RawCdpCookie[], targetUrl: string): MapCookiesResult;
154
+ export interface MapOriginsResult {
155
+ origins: OriginStorage[];
156
+ totalSeen: number;
157
+ droppedOutOfScope: number;
158
+ }
159
+ /**
160
+ * @desc Scope raw per-origin localStorage dumps down to origins that are
161
+ * `targetUrl`'s own host or a subdomain of it (`isOriginHostInScope`).
162
+ * `origin` is passed through unchanged (it is already a bare
163
+ * `scheme://host[:port]` by construction — callers build it via
164
+ * `new URL(pageUrl).origin`) so it satisfies sessionShape.js's bare-origin
165
+ * check untouched.
166
+ */
167
+ export declare function mapOrigins(rawOrigins: RawOriginLocalStorage[], targetUrl: string): MapOriginsResult;
@@ -0,0 +1,227 @@
1
+ /**
2
+ * Pure CDP -> storageState mapping + domain scoping for `scraps account
3
+ * session capture` (trawl_cli#183). No I/O here — every function takes
4
+ * plain data and returns plain data, so the full cookie/localStorage
5
+ * matrix (sameSite casing, expires-in-seconds, domain scoping, …) is
6
+ * testable without a real Chrome.
7
+ *
8
+ * The output shape mirrors trawl_node#1976's ingest validator
9
+ * (modules/scraps/helpers/sessionShape.js) — CookieData fields, `sameSite`
10
+ * in `Strict|Lax|None`, `expires` in seconds, `domain` required. This file
11
+ * pre-normalizes to that shape so the server's validator never has to
12
+ * reject the WHOLE batch over one anomalous cookie the client could have
13
+ * caught itself (see normalizeCdpCookie's sameSite=None handling below).
14
+ */
15
+ /**
16
+ * Domain scoping (trawl_cli#183 review finding 2 — cross-tenant leak). The
17
+ * previous version of this file peeled a hostname down to a "registrable
18
+ * domain" (last 2-3 labels, with a small hardcoded allowlist for
19
+ * `co.uk`-shaped suffixes) and treated any host sharing that suffix as in
20
+ * scope. That is wrong for any multi-tenant hosting domain NOT in the
21
+ * allowlist — `getRegistrableDomain('alice.github.io')` collapsed to
22
+ * `'github.io'`, so `isHostInScope('bob.github.io', 'github.io')` came back
23
+ * `true`: a scrap targeting `alice.github.io` would upload `bob.github.io`'s
24
+ * cookies too (same for `herokuapp.com`, `vercel.app`, `s3.amazonaws.com`,
25
+ * …) — every one of them would need its own allowlist entry, which is
26
+ * exactly the Public Suffix List this file deliberately avoids depending
27
+ * on.
28
+ *
29
+ * The fix NARROWS instead: scope every capture to RFC 6265 §5.1.3
30
+ * domain-matching against the scrap URL's own host (`getTargetHost`) —
31
+ * never a peeled/derived domain, so there is nothing left to get wrong
32
+ * about a given hosting provider's suffix.
33
+ *
34
+ * Verified against a real Chrome 152 (loopback probe, `--host-resolver-rules`
35
+ * remapping `alice.github.io`/`app.example.com`/etc. to 127.0.0.1, no real
36
+ * network traffic):
37
+ * - Chrome REFUSES to let a page at `alice.github.io` set a cookie with
38
+ * `Domain=github.io` at all — it never reaches the cookie jar. That's
39
+ * Chrome's own PSL enforcement on the SET side, and it's what makes the
40
+ * zero-PSL-dependency approach below safe: a public-suffix-scoped cookie
41
+ * simply cannot exist in a real jar to be mis-scoped in the first place.
42
+ * - `Storage.getCookies` DOES distinguish a host-only cookie from an
43
+ * explicit-`Domain=` one at the CDP layer, even when the explicit domain
44
+ * string equals the current host exactly: a bare `document.cookie="k=v"`
45
+ * on `alice.github.io` comes back with `domain:"alice.github.io"` (no
46
+ * leading dot); `document.cookie="k=v; domain=alice.github.io"` — same
47
+ * effective host — comes back `domain:".alice.github.io"` (leading dot).
48
+ * That leading dot is real, load-bearing information: it is exactly
49
+ * RFC 6265's own signal for "this is a domain-match cookie, not a
50
+ * host-only one" and `isCookieDomainInScope` below reads it as such
51
+ * rather than stripping it.
52
+ *
53
+ * One deliberate loss, not a bug: a host-only cookie set on a SIBLING host
54
+ * (e.g. `auth.example.com` while the scrap targets `app.example.com`) is
55
+ * out of scope here. A real browser would never send that cookie to
56
+ * `app.example.com` either — host-only means exactly that host — so
57
+ * nothing replay-relevant is lost; it would only matter if the scrap
58
+ * script itself later navigates to `auth.example.com`, which this capture
59
+ * has no visibility into anyway.
60
+ */
61
+ /**
62
+ * @desc The exact host `targetUrl` will actually be requested at — the
63
+ * single anchor every scope decision in this file is relative to. NOT a
64
+ * peeled "registrable domain": no suffix list, no label-counting, no
65
+ * allowlist to keep up to date.
66
+ */
67
+ export function getTargetHost(targetUrl) {
68
+ return new URL(targetUrl).hostname.toLowerCase();
69
+ }
70
+ /**
71
+ * @desc RFC 6265 §5.1.3 domain-matching for one captured cookie's `domain`
72
+ * attribute exactly as CDP's `Storage.getCookies` reports it (leading dot
73
+ * preserved when Chrome ever wrote one, absent for a host-only cookie —
74
+ * see the module doc comment for how that was verified against a real
75
+ * Chrome).
76
+ * - Leading dot (a domain-match cookie): in scope for the target host
77
+ * itself, or any subdomain of it.
78
+ * - No leading dot (a host-only cookie): in scope ONLY for that exact
79
+ * host — never a subdomain, never a sibling, never a parent. This is
80
+ * what excludes `bob.github.io` from a capture targeting
81
+ * `alice.github.io`, and `auth.example.com` from one targeting
82
+ * `app.example.com`.
83
+ */
84
+ export function isCookieDomainInScope(rawDomain, targetHost) {
85
+ const host = targetHost.toLowerCase();
86
+ if (rawDomain.startsWith('.')) {
87
+ const domain = rawDomain.slice(1).toLowerCase();
88
+ return host === domain || host.endsWith(`.${domain}`);
89
+ }
90
+ return host === rawDomain.toLowerCase();
91
+ }
92
+ /**
93
+ * @desc True when a captured page's own hostname (`originHost` — an
94
+ * origin's localStorage has no "domain attribute" concept, unlike a
95
+ * cookie) is the target host itself, or a subdomain of it. Anchored the
96
+ * opposite way from a host-only cookie check: here the TARGET is the root
97
+ * and the origin must fall under it, so a sibling (`auth.example.com` for
98
+ * a scrap on `app.example.com`) is still excluded, but a genuine
99
+ * subdomain page opened during the same login flow is not.
100
+ */
101
+ export function isOriginHostInScope(originHost, targetHost) {
102
+ const h = originHost.toLowerCase();
103
+ const target = targetHost.toLowerCase();
104
+ return h === target || h.endsWith(`.${target}`);
105
+ }
106
+ const CANONICAL_SAME_SITE = {
107
+ strict: 'Strict',
108
+ lax: 'Lax',
109
+ none: 'None',
110
+ };
111
+ /**
112
+ * @desc Recase any CDP/browser sameSite spelling onto CookieData's exact
113
+ * casing. Returns null for anything unrecognized (CDP's own "unspecified"
114
+ * included) — the field is then omitted entirely rather than guessed,
115
+ * mirroring sessionShape.js's own 'unspecified' handling.
116
+ */
117
+ function canonicalizeSameSite(value) {
118
+ return CANONICAL_SAME_SITE[value.toLowerCase()] ?? null;
119
+ }
120
+ /**
121
+ * @desc Map one CDP cookie onto CookieData, or null if it's structurally
122
+ * unusable (no domain, no name/value). A `sameSite=None` cookie missing
123
+ * `secure` is NOT dropped outright — dropping only the `sameSite`
124
+ * attribute (rather than the whole cookie) is deliberate: sessionShape.js's
125
+ * `normalizeCookies` throws on that combination, and since it processes
126
+ * the array with a single `.map()`, ONE such cookie would reject the
127
+ * ENTIRE upload with a 422 — a capture that otherwise succeeded would be
128
+ * thrown away over one anomalous cookie. A real, currently-live browser
129
+ * cookie can't actually be in this state (Chrome refuses to store a `None`
130
+ * cookie without `Secure`), so in practice this only guards a defensive
131
+ * edge; when it does happen, omitting `sameSite` keeps the cookie (and the
132
+ * rest of the batch) valid.
133
+ */
134
+ function normalizeCdpCookie(raw) {
135
+ if (typeof raw.name !== 'string' || raw.name.length === 0)
136
+ return null;
137
+ if (typeof raw.value !== 'string')
138
+ return null;
139
+ if (typeof raw.domain !== 'string' || raw.domain.length === 0)
140
+ return null;
141
+ const cookie = { name: raw.name, value: raw.value, domain: raw.domain };
142
+ if (typeof raw.path === 'string')
143
+ cookie.path = raw.path;
144
+ if (typeof raw.secure === 'boolean')
145
+ cookie.secure = raw.secure;
146
+ if (typeof raw.httpOnly === 'boolean')
147
+ cookie.httpOnly = raw.httpOnly;
148
+ if (typeof raw.sameSite === 'string' && raw.sameSite.length > 0) {
149
+ const canon = canonicalizeSameSite(raw.sameSite);
150
+ if (canon && !(canon === 'None' && cookie.secure !== true)) {
151
+ cookie.sameSite = canon;
152
+ }
153
+ }
154
+ if (typeof raw.expires === 'number' && !Number.isNaN(raw.expires) && raw.expires !== -1) {
155
+ cookie.expires = raw.expires;
156
+ }
157
+ return cookie;
158
+ }
159
+ /**
160
+ * @desc Scope + normalize a raw CDP cookie jar down to the cookies a real
161
+ * browser would actually send to `targetUrl`'s host (`isCookieDomainInScope`
162
+ * — RFC 6265 domain-matching, never a peeled registrable domain). Never
163
+ * throws — a structurally broken cookie is counted and dropped, not fatal
164
+ * to the rest of the capture (see the "batch same class failures" rule:
165
+ * this IS the sweep of the offending class, applied once here rather than
166
+ * duplicated at each call site).
167
+ */
168
+ export function mapCookies(rawCookies, targetUrl) {
169
+ const targetHost = getTargetHost(targetUrl);
170
+ let droppedOutOfScope = 0;
171
+ let droppedInvalid = 0;
172
+ const cookies = [];
173
+ for (const raw of rawCookies) {
174
+ if (!raw || typeof raw.domain !== 'string' || raw.domain.length === 0) {
175
+ droppedInvalid++;
176
+ continue;
177
+ }
178
+ // The leading '.' (or its absence) is meaningful — see
179
+ // isCookieDomainInScope's own doc comment — so it is NOT stripped here;
180
+ // only the OUTPUT cookie's `domain` field stays untouched either way
181
+ // (normalizeCdpCookie passes raw.domain through as-is, dot included),
182
+ // since the server needs the exact original attribute to replay it.
183
+ if (!isCookieDomainInScope(raw.domain, targetHost)) {
184
+ droppedOutOfScope++;
185
+ continue;
186
+ }
187
+ const mapped = normalizeCdpCookie(raw);
188
+ if (!mapped) {
189
+ droppedInvalid++;
190
+ continue;
191
+ }
192
+ cookies.push(mapped);
193
+ }
194
+ return { cookies, totalSeen: rawCookies.length, droppedOutOfScope, droppedInvalid };
195
+ }
196
+ /**
197
+ * @desc Scope raw per-origin localStorage dumps down to origins that are
198
+ * `targetUrl`'s own host or a subdomain of it (`isOriginHostInScope`).
199
+ * `origin` is passed through unchanged (it is already a bare
200
+ * `scheme://host[:port]` by construction — callers build it via
201
+ * `new URL(pageUrl).origin`) so it satisfies sessionShape.js's bare-origin
202
+ * check untouched.
203
+ */
204
+ export function mapOrigins(rawOrigins, targetUrl) {
205
+ const targetHost = getTargetHost(targetUrl);
206
+ let droppedOutOfScope = 0;
207
+ const origins = [];
208
+ for (const entry of rawOrigins) {
209
+ let host;
210
+ try {
211
+ host = new URL(entry.origin).hostname;
212
+ }
213
+ catch {
214
+ droppedOutOfScope++;
215
+ continue;
216
+ }
217
+ if (!isOriginHostInScope(host, targetHost)) {
218
+ droppedOutOfScope++;
219
+ continue;
220
+ }
221
+ const localStorage = entry.entries
222
+ .filter((kv) => typeof kv?.name === 'string' && typeof kv?.value === 'string')
223
+ .map((kv) => ({ name: kv.name, value: kv.value }));
224
+ origins.push({ origin: entry.origin, localStorage });
225
+ }
226
+ return { origins, totalSeen: rawOrigins.length, droppedOutOfScope };
227
+ }
@@ -166,7 +166,15 @@ just to look helpful, so don't treat its absence as an error.
166
166
 
167
167
  For the full, versioned machine description of every command (arguments,
168
168
  options, aliases, exit codes, error kinds) — instead of parsing this doc —
169
- run `trawl spec --json`.
169
+ run `trawl spec --json`. It also carries `docsUrl` (resolved from the server
170
+ when declared, else derived from a known first-party API host, else omitted
171
+ — never a guessed URL) and, where a guide exists, a per-command `docs` deep
172
+ link — but the deep link and `llmsUrl` are narrower than `docsUrl`: they only
173
+ ever come from the known-first-party-host derivation, never from a
174
+ server-declared `docsUrl`, since this CLI's own guide slugs have no reason
175
+ to exist on a third party's own docs root. A `failureKind:"auth"` run's
176
+ payload (`doctor`/`data --errors`/`run-info`) carries the same `docs` field,
177
+ under the same rule.
170
178
 
171
179
  ## Non-interactive contract
172
180
 
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@trawlme/cli",
3
- "version": "3.10.0",
3
+ "version": "3.12.0",
4
4
  "description": "Trawl CLI — manage scraps from the terminal",
5
5
  "type": "module",
6
6
  "bin": {