@trawlme/cli 3.10.0 → 3.12.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,122 @@
1
+ /**
2
+ * Spawn a headed, isolated Chrome instance wired for raw CDP over
3
+ * `--remote-debugging-pipe` (trawl_cli#183). Thin glue only — the actual
4
+ * protocol lives in cdp-pipe.ts, kept separate so this file's job (build
5
+ * the right flags, wire fd 3/4, fail loudly if the platform doesn't expose
6
+ * them) stays small enough to unit test with a mocked `spawn`.
7
+ */
8
+ import { spawn } from 'node:child_process';
9
+ import { CdpPipe } from './cdp-pipe.js';
10
+ /**
11
+ * @desc The exact CLI flags Chrome is launched with. Pure + exported so
12
+ * the flag set is testable without spawning a real process.
13
+ * - `--remote-debugging-pipe` — CDP over fd 3 (in) / fd 4 (out), the
14
+ * zero-dependency transport this command relies on.
15
+ * - `--user-data-dir` — an ephemeral, isolated profile (never the user's
16
+ * real Chrome profile/cookies).
17
+ * - `--no-first-run` / `--no-default-browser-check` — skip first-run UI
18
+ * that would otherwise sit in front of the target URL.
19
+ * - `--disable-sync` — never touches the user's real Google account sync.
20
+ * - `--password-store=basic` / `--use-mock-keychain` — a fresh profile
21
+ * would otherwise prompt for the macOS login keychain (Safe Storage) the
22
+ * first time Chrome touches its cookie/password store; these keep the
23
+ * whole flow non-blocking on an ephemeral profile that's deleted right
24
+ * after (mirrors Crawl4AI's `crwl profiles`, cited in the issue).
25
+ * - `--new-window` — always its own window, never a tab folded into an
26
+ * already-running Chrome instance.
27
+ * - No `--no-sandbox`: this runs headed on the user's own machine, not in
28
+ * a locked-down CI container — the sandbox stays on.
29
+ */
30
+ export function buildChromeArgs(userDataDir, targetUrl) {
31
+ return [
32
+ '--remote-debugging-pipe',
33
+ `--user-data-dir=${userDataDir}`,
34
+ '--no-first-run',
35
+ '--no-default-browser-check',
36
+ '--disable-sync',
37
+ '--password-store=basic',
38
+ '--use-mock-keychain',
39
+ '--new-window',
40
+ targetUrl,
41
+ ];
42
+ }
43
+ /**
44
+ * @desc Launch Chrome and wire a CdpPipe to its fd 3 (write) / fd 4 (read).
45
+ * Returns a Promise rather than the `ChildProcess` synchronously: `spawn()`
46
+ * can fail AFTER returning — EACCES on a non-executable path, an ENOENT
47
+ * race, EPERM, E2BIG — by emitting an `'error'` event rather than throwing,
48
+ * and that event is attached to BEFORE the fd 3/4 check below so a failure
49
+ * from either cause rejects this promise instead of crashing the process
50
+ * (an EventEmitter with zero `'error'` listeners throws the error itself).
51
+ * @rejects {Error} if `spawn()` itself failed (the original error, so
52
+ * `.code` like `EACCES` survives), or if the platform/Chrome build didn't
53
+ * expose fd 3/4 as pipes (e.g. `--remote-debugging-pipe` unsupported) — a
54
+ * factual message, never a silent hang.
55
+ */
56
+ export function launchChrome(executablePath, targetUrl, userDataDir, spawnFn = spawn) {
57
+ return new Promise((resolve, reject) => {
58
+ // stderr is 'ignore', not 'pipe': nothing here ever reads it (unlike the
59
+ // rejected `ws`-over-port design, `--remote-debugging-pipe` needs none of
60
+ // Chrome's stderr output), and a 'pipe' nobody drains fills its OS pipe
61
+ // buffer — Chrome's own logging would then block on write mid-session,
62
+ // freezing the browser the human is mid-login in.
63
+ const proc = spawnFn(executablePath, buildChromeArgs(userDataDir, targetUrl), {
64
+ stdio: ['ignore', 'ignore', 'ignore', 'pipe', 'pipe'],
65
+ // POSIX only: makes Chrome the leader of its OWN process group, so its
66
+ // helper subprocesses (renderer/GPU/network service — Chrome forks
67
+ // several even for one window) live in that group too, distinct from
68
+ // ours. Verified against a real launch (trawl_cli#183's loopback E2E):
69
+ // without this, `proc.kill('SIGKILL')` only killed the main Chrome
70
+ // process — its helpers briefly kept the user-data-dir's files open,
71
+ // and `rmSync` silently lost that race, leaking the temp profile on
72
+ // every run. session-capture.ts's cleanup kills `-proc.pid` (the whole
73
+ // group) instead of `proc.pid` alone, which requires this.
74
+ detached: process.platform !== 'win32',
75
+ });
76
+ let settled = false;
77
+ // Attached as the very first thing after `spawnFn` returns — before the
78
+ // fd 3/4 check below, and before anything here ever awaits — so nothing
79
+ // can slip through unobserved. Stays attached FOREVER, including after
80
+ // this promise settles: a LATE error (spawn succeeded, then something
81
+ // else goes wrong long after this promise already resolved) must not
82
+ // crash the process either, and once `settled` is true this listener is
83
+ // a permanent no-op — whatever consumes `proc` from here on notices
84
+ // Chrome went away some other way (the CDP pipe closing).
85
+ proc.on('error', (err) => {
86
+ if (settled)
87
+ return;
88
+ settled = true;
89
+ try {
90
+ proc.kill('SIGKILL');
91
+ }
92
+ catch {
93
+ // already gone
94
+ }
95
+ reject(err);
96
+ });
97
+ const writeStream = proc.stdio[3];
98
+ const readStream = proc.stdio[4];
99
+ if (!writeStream || !readStream || typeof writeStream.write !== 'function') {
100
+ settled = true;
101
+ try {
102
+ proc.kill('SIGKILL');
103
+ }
104
+ catch {
105
+ // already gone
106
+ }
107
+ reject(new Error('Chrome did not expose the CDP pipe (fd 3/4) — this Chrome build or platform may not support --remote-debugging-pipe.'));
108
+ return;
109
+ }
110
+ // Only declare success once the OS confirms the process actually
111
+ // started — the documented signal for that ('spawn', not merely
112
+ // "spawn() returned without throwing") — rather than assuming
113
+ // immediately: that's exactly the assumption the bug this fixes was
114
+ // built on.
115
+ proc.once('spawn', () => {
116
+ if (settled)
117
+ return;
118
+ settled = true;
119
+ resolve({ proc, cdp: new CdpPipe(writeStream, readStream) });
120
+ });
121
+ });
122
+ }
@@ -0,0 +1,140 @@
1
+ /**
2
+ * trawl_cli#185 — ONE source of truth for the docs URL(s) an agent (or
3
+ * human) can be pointed at from three surfaces: `spec --json` (top-level
4
+ * `docsUrl`/`llmsUrl` + a per-command `docs` deep link), a run's JSON
5
+ * payload (`doctor`/`data --errors`/`run-info`, keyed on `failureKind` —
6
+ * NOT errors.ts's unrelated `ErrorEnvelope.kind`, see the doc comment on
7
+ * `FAILURE_KIND_DOC_PATHS` below), and the `--help` footer. Three hardcoded
8
+ * lists here would diverge within months — that exact "same fact computed
9
+ * in two places" defect has recurred repeatedly across this epic — so every
10
+ * surface reads these same tables/functions, never its own copy.
11
+ *
12
+ * Resolution ladder (issue #185 — all three rungs REQUIRED, in this order,
13
+ * evaluated PER FIELD, never as one bundled decision — see `resolveDocsUrls`):
14
+ *
15
+ * 1. Prefer the server. `externalDocs.url` on the OpenAPI document at
16
+ * `<apiBase>/api/spec.json` is the standard OpenAPI field for exactly
17
+ * this, and reading it makes a self-hosted install work with ZERO CLI
18
+ * change. It is unset (null) server-side today, so this is
19
+ * forward-looking — build it anyway, and it must win outright over rung
20
+ * 2 when present. The one caller allowed to fetch it is `spec.ts`'s own
21
+ * action, bounded and swallowed-on-failure (mirrors lib/tips.ts's
22
+ * `isReferralProgramUserFacing`) — a deliberate, once-per-invocation
23
+ * agent probe, not "every command". Every other surface (an error
24
+ * payload on an arbitrary failing command, the `--help` footer) must
25
+ * never add a network call or a failure mode to a command nobody asked
26
+ * to hit the docs host for — see `resolveDocsUrls`'s `externalDocsUrl`
27
+ * parameter, which only ever arrives pre-fetched.
28
+ * 2. Else derive, by stripping a leading `api.` host label from the
29
+ * configured API base — but ONLY for a KNOWN first-party `trawl.me`
30
+ * host. `api.trawl.me` -> `trawl.me` (prod: API and docs are genuinely
31
+ * on different hosts); `dev.trawl.me` (no `api.` prefix) -> unchanged
32
+ * (dev serves docs on the SAME host as its API). A generic, unscoped
33
+ * `api.`-strip applied to ANY host is itself a guess — it assumes a
34
+ * self-hosted `api.acme.internal` serves docs at `acme.internal`, which
35
+ * nothing here can know. Scoping the strip to `trawl.me` is what makes
36
+ * rung 3 (below) ever actually fire for a self-hosted base.
37
+ * 3. Else OMIT the field entirely. Never emit a guessed URL (issue's Rule
38
+ * 3): a wrong URL sends an agent to a 404 WITH CONFIDENCE, worse than no
39
+ * URL at all — a self-hosted install with no server-declared
40
+ * `externalDocs` and a base outside `trawl.me` gets no `docsUrl`/
41
+ * `llmsUrl`, not a hopeful default pointed at OUR docs host.
42
+ *
43
+ * `llmsUrl` has no OpenAPI-standard field to read (rung 1 contributes
44
+ * nothing to it) — deriving it from the ORIGIN of a server-declared
45
+ * `externalDocs.url` would itself be a guess for a self-hosted install
46
+ * (rule 3 again), so `llmsUrl` resolves ONLY via rung 2 (the known-host
47
+ * derivation) or omission. It never rides along with a rung-1 `docsUrl`.
48
+ *
49
+ * PROVENANCE (defect fix, reviewer repro against a live mock server): the
50
+ * top-level `docsUrl` above is deliberately the FLATTENED "whichever rung
51
+ * won" value — right for a human reading `spec --json`'s top-level field,
52
+ * WRONG as an input to `resolveCommandDocsUrl`/`resolveFailureKindDocsUrl`
53
+ * below. Those two append one of THIS CLI's own hardcoded guide slugs
54
+ * (`COMMAND_DOC_PATHS`/`FAILURE_KIND_DOC_PATHS`) onto whatever `docsUrl`
55
+ * resolved — safe onto a rung-2 root we derived ourselves (we know
56
+ * trawl.me's guide tree), never safe onto a rung-1 root (an arbitrary
57
+ * third party's own docs site, self-hosted, with no reason to carry our
58
+ * slugs). A mock server declaring `externalDocs.url: 'http://h/guide'`
59
+ * used to get `http://h/guide/build-your-scrap/account-sessions` appended
60
+ * — a confidently-wrong 404. So `DocsUrls.docsUrlIsDerived` carries rung
61
+ * provenance ALONGSIDE the string (never flattened to a bare string again)
62
+ * and both deep-link functions now take the whole `DocsUrls`-shaped object
63
+ * and gate construction on that flag — see its own doc comment below.
64
+ */
65
+ export interface DocsUrls {
66
+ docsUrl?: string;
67
+ llmsUrl?: string;
68
+ /**
69
+ * True iff `docsUrl` came from rung 2 (THIS CLI's own derivation from a
70
+ * KNOWN `trawl.me` host) — the only case where it is safe to append one of
71
+ * this CLI's own guide slugs onto it (see `resolveCommandDocsUrl`/
72
+ * `resolveFailureKindDocsUrl`). False when `docsUrl` is a rung-1
73
+ * SERVER-declared root instead — an arbitrary third party's own docs site,
74
+ * which has no reason to host our guide slugs, even when that server
75
+ * happens to run on a `trawl.me` host. Always a defined boolean whenever
76
+ * `docsUrl` itself is present; undefined only when `docsUrl` is undefined
77
+ * too (rung 3 — nothing resolved at all).
78
+ */
79
+ docsUrlIsDerived?: boolean;
80
+ }
81
+ /**
82
+ * Rung 2 — see the module doc comment. Returns `{ protocol, host }` for a
83
+ * KNOWN first-party API base (preserving the base's own protocol rather
84
+ * than assuming `https:`), or `null` when the base isn't recognized
85
+ * (self-hosted, a custom domain, `localhost`, an unparseable string, …) —
86
+ * `null` must propagate to omission (rung 3), never to a guessed host.
87
+ */
88
+ export declare function deriveDocsOrigin(apiBaseUrl: string): {
89
+ protocol: string;
90
+ host: string;
91
+ } | null;
92
+ /** Same rung as `deriveDocsOrigin`, collapsed to just the host string —
93
+ * convenience for a caller that only needs the derived host, not the
94
+ * protocol (kept as its own export since `docs.test.ts` exercises the host
95
+ * derivation independently of protocol handling). */
96
+ export declare function deriveDocsHost(apiBaseUrl: string): string | null;
97
+ /**
98
+ * The full ladder, PER FIELD (see module doc comment for why `llmsUrl`
99
+ * cannot ride along with a rung-1 `docsUrl`). Pure and synchronous — no
100
+ * network, safe to call from any surface (an error payload, the `--help`
101
+ * footer) without adding latency or a new failure mode. `externalDocsUrl`
102
+ * is rung 1's input: only ever supplied by `spec.ts`'s action, after its
103
+ * own bounded, swallowed-on-failure fetch — every other caller omits it and
104
+ * gets rungs 2/3 only.
105
+ */
106
+ export declare function resolveDocsUrls(opts: {
107
+ apiBaseUrl: string;
108
+ externalDocsUrl?: string | null;
109
+ }): DocsUrls;
110
+ /**
111
+ * `resolveCommandDocsUrl` returns `undefined` (never a bare path or a
112
+ * relative link) whenever any of three things is missing — `docsUrl`
113
+ * unresolved (rung 3 already fired), `docsUrl` resolved but NOT derived
114
+ * (rung 1 — a server-declared root; see `DocsUrls.docsUrlIsDerived`'s doc
115
+ * comment for why appending our own guide slug onto a third party's root is
116
+ * exactly the confidently-wrong-404 defect this gate closes), or no guide is
117
+ * mapped for this command — so a spec consumer never has to special-case a
118
+ * partial value. Takes the whole resolved `DocsUrls` object (never a bare
119
+ * string) specifically so this provenance can never again be flattened away
120
+ * before it reaches here.
121
+ */
122
+ export declare function resolveCommandDocsUrl(commandName: string, docs: Pick<DocsUrls, 'docsUrl' | 'docsUrlIsDerived'>): string | undefined;
123
+ /** Same "undefined unless docsUrl is resolved AND derived (rung 2)" gate as
124
+ * `resolveCommandDocsUrl` above, same reason — never append this CLI's own
125
+ * guide slug onto a rung-1 server-declared root. Callers MUST gate the call
126
+ * on their own staleness rule first (e.g. `doctor.ts`'s `isAuthWall`) — this
127
+ * function only knows the string-to-path mapping, not whether a stamped
128
+ * `failureKind` is still live on a run patched afterward. */
129
+ export declare function resolveFailureKindDocsUrl(kind: string | null | undefined, docs: Pick<DocsUrls, 'docsUrl' | 'docsUrlIsDerived'>): string | undefined;
130
+ /**
131
+ * `--help` footer text (issue #185 scope item 3) — a single dim line, for
132
+ * humans, e.g. "Docs: https://trawl.me/docs". Returns `undefined` (never an
133
+ * empty or guessed line) when no `docsUrl` resolved — mirrors
134
+ * lib/tips.ts's shape (a pure function separated from IO, so it's testable
135
+ * without mocking chalk/console) but NOT its TTY gate: tips.ts suppresses a
136
+ * promotional nudge from piped output, but a docs line inside `--help |
137
+ * less` is exactly the kind of thing worth keeping. Un-colored here — the
138
+ * caller (index.ts) applies `chalk.dim` so this stays trivially testable.
139
+ */
140
+ export declare function docsFooterLine(docsUrl?: string): string | undefined;
@@ -0,0 +1,238 @@
1
+ /**
2
+ * trawl_cli#185 — ONE source of truth for the docs URL(s) an agent (or
3
+ * human) can be pointed at from three surfaces: `spec --json` (top-level
4
+ * `docsUrl`/`llmsUrl` + a per-command `docs` deep link), a run's JSON
5
+ * payload (`doctor`/`data --errors`/`run-info`, keyed on `failureKind` —
6
+ * NOT errors.ts's unrelated `ErrorEnvelope.kind`, see the doc comment on
7
+ * `FAILURE_KIND_DOC_PATHS` below), and the `--help` footer. Three hardcoded
8
+ * lists here would diverge within months — that exact "same fact computed
9
+ * in two places" defect has recurred repeatedly across this epic — so every
10
+ * surface reads these same tables/functions, never its own copy.
11
+ *
12
+ * Resolution ladder (issue #185 — all three rungs REQUIRED, in this order,
13
+ * evaluated PER FIELD, never as one bundled decision — see `resolveDocsUrls`):
14
+ *
15
+ * 1. Prefer the server. `externalDocs.url` on the OpenAPI document at
16
+ * `<apiBase>/api/spec.json` is the standard OpenAPI field for exactly
17
+ * this, and reading it makes a self-hosted install work with ZERO CLI
18
+ * change. It is unset (null) server-side today, so this is
19
+ * forward-looking — build it anyway, and it must win outright over rung
20
+ * 2 when present. The one caller allowed to fetch it is `spec.ts`'s own
21
+ * action, bounded and swallowed-on-failure (mirrors lib/tips.ts's
22
+ * `isReferralProgramUserFacing`) — a deliberate, once-per-invocation
23
+ * agent probe, not "every command". Every other surface (an error
24
+ * payload on an arbitrary failing command, the `--help` footer) must
25
+ * never add a network call or a failure mode to a command nobody asked
26
+ * to hit the docs host for — see `resolveDocsUrls`'s `externalDocsUrl`
27
+ * parameter, which only ever arrives pre-fetched.
28
+ * 2. Else derive, by stripping a leading `api.` host label from the
29
+ * configured API base — but ONLY for a KNOWN first-party `trawl.me`
30
+ * host. `api.trawl.me` -> `trawl.me` (prod: API and docs are genuinely
31
+ * on different hosts); `dev.trawl.me` (no `api.` prefix) -> unchanged
32
+ * (dev serves docs on the SAME host as its API). A generic, unscoped
33
+ * `api.`-strip applied to ANY host is itself a guess — it assumes a
34
+ * self-hosted `api.acme.internal` serves docs at `acme.internal`, which
35
+ * nothing here can know. Scoping the strip to `trawl.me` is what makes
36
+ * rung 3 (below) ever actually fire for a self-hosted base.
37
+ * 3. Else OMIT the field entirely. Never emit a guessed URL (issue's Rule
38
+ * 3): a wrong URL sends an agent to a 404 WITH CONFIDENCE, worse than no
39
+ * URL at all — a self-hosted install with no server-declared
40
+ * `externalDocs` and a base outside `trawl.me` gets no `docsUrl`/
41
+ * `llmsUrl`, not a hopeful default pointed at OUR docs host.
42
+ *
43
+ * `llmsUrl` has no OpenAPI-standard field to read (rung 1 contributes
44
+ * nothing to it) — deriving it from the ORIGIN of a server-declared
45
+ * `externalDocs.url` would itself be a guess for a self-hosted install
46
+ * (rule 3 again), so `llmsUrl` resolves ONLY via rung 2 (the known-host
47
+ * derivation) or omission. It never rides along with a rung-1 `docsUrl`.
48
+ *
49
+ * PROVENANCE (defect fix, reviewer repro against a live mock server): the
50
+ * top-level `docsUrl` above is deliberately the FLATTENED "whichever rung
51
+ * won" value — right for a human reading `spec --json`'s top-level field,
52
+ * WRONG as an input to `resolveCommandDocsUrl`/`resolveFailureKindDocsUrl`
53
+ * below. Those two append one of THIS CLI's own hardcoded guide slugs
54
+ * (`COMMAND_DOC_PATHS`/`FAILURE_KIND_DOC_PATHS`) onto whatever `docsUrl`
55
+ * resolved — safe onto a rung-2 root we derived ourselves (we know
56
+ * trawl.me's guide tree), never safe onto a rung-1 root (an arbitrary
57
+ * third party's own docs site, self-hosted, with no reason to carry our
58
+ * slugs). A mock server declaring `externalDocs.url: 'http://h/guide'`
59
+ * used to get `http://h/guide/build-your-scrap/account-sessions` appended
60
+ * — a confidently-wrong 404. So `DocsUrls.docsUrlIsDerived` carries rung
61
+ * provenance ALONGSIDE the string (never flattened to a bare string again)
62
+ * and both deep-link functions now take the whole `DocsUrls`-shaped object
63
+ * and gate construction on that flag — see its own doc comment below.
64
+ */
65
+ /** The one first-party domain this CLI knows to serve docs — see rung 2 in
66
+ * the module doc comment above. Deliberately not "any host", so an
67
+ * unrecognized (self-hosted/custom) base falls through to omission. */
68
+ const KNOWN_DOCS_DOMAIN = 'trawl.me';
69
+ /**
70
+ * Rung 2 — see the module doc comment. Returns `{ protocol, host }` for a
71
+ * KNOWN first-party API base (preserving the base's own protocol rather
72
+ * than assuming `https:`), or `null` when the base isn't recognized
73
+ * (self-hosted, a custom domain, `localhost`, an unparseable string, …) —
74
+ * `null` must propagate to omission (rung 3), never to a guessed host.
75
+ */
76
+ export function deriveDocsOrigin(apiBaseUrl) {
77
+ let url;
78
+ try {
79
+ url = new URL(apiBaseUrl);
80
+ }
81
+ catch {
82
+ return null;
83
+ }
84
+ const hostname = url.hostname.toLowerCase();
85
+ const stripped = hostname.startsWith('api.') ? hostname.slice('api.'.length) : hostname;
86
+ if (stripped === KNOWN_DOCS_DOMAIN || stripped.endsWith(`.${KNOWN_DOCS_DOMAIN}`)) {
87
+ return { protocol: url.protocol, host: stripped };
88
+ }
89
+ return null;
90
+ }
91
+ /** Same rung as `deriveDocsOrigin`, collapsed to just the host string —
92
+ * convenience for a caller that only needs the derived host, not the
93
+ * protocol (kept as its own export since `docs.test.ts` exercises the host
94
+ * derivation independently of protocol handling). */
95
+ export function deriveDocsHost(apiBaseUrl) {
96
+ return deriveDocsOrigin(apiBaseUrl)?.host ?? null;
97
+ }
98
+ /**
99
+ * The full ladder, PER FIELD (see module doc comment for why `llmsUrl`
100
+ * cannot ride along with a rung-1 `docsUrl`). Pure and synchronous — no
101
+ * network, safe to call from any surface (an error payload, the `--help`
102
+ * footer) without adding latency or a new failure mode. `externalDocsUrl`
103
+ * is rung 1's input: only ever supplied by `spec.ts`'s action, after its
104
+ * own bounded, swallowed-on-failure fetch — every other caller omits it and
105
+ * gets rungs 2/3 only.
106
+ */
107
+ export function resolveDocsUrls(opts) {
108
+ const origin = deriveDocsOrigin(opts.apiBaseUrl);
109
+ const derived = origin
110
+ ? {
111
+ docsUrl: `${origin.protocol}//${origin.host}/docs`,
112
+ llmsUrl: `${origin.protocol}//${origin.host}/llms.txt`,
113
+ docsUrlIsDerived: true,
114
+ }
115
+ : {};
116
+ if (opts.externalDocsUrl) {
117
+ // Rung 1 wins for docsUrl outright; llmsUrl stays rung-2-only (see doc
118
+ // comment) — a self-hosted `externalDocsUrl` outside `trawl.me` yields
119
+ // `{ docsUrl: <server's own> }` with `llmsUrl` omitted, never guessed.
120
+ // `docsUrlIsDerived: false` regardless of whether rung 2 ALSO resolved
121
+ // (e.g. a `trawl.me`-hosted server declaring its own externalDocs) — a
122
+ // server-declared root is never "derived" by THIS CLI, and never a safe
123
+ // base for our own guide slugs (see DocsUrls.docsUrlIsDerived).
124
+ return derived.llmsUrl
125
+ ? { docsUrl: opts.externalDocsUrl, llmsUrl: derived.llmsUrl, docsUrlIsDerived: false }
126
+ : { docsUrl: opts.externalDocsUrl, docsUrlIsDerived: false };
127
+ }
128
+ return derived;
129
+ }
130
+ /** Join a docs root (e.g. `https://trawl.me/docs`) with a guide path
131
+ * suffix (e.g. `build-your-scrap/account-sessions`) — plain string
132
+ * concatenation, deliberately NOT `new URL(suffix, docsUrl)`: a
133
+ * leading-slash suffix resolved against a URL with its own path component
134
+ * (`/docs`) would resolve relative to the ORIGIN, silently dropping
135
+ * `/docs` from the result (`new URL('/x', 'https://h/docs')` ==
136
+ * `https://h/x`, not `https://h/docs/x`). */
137
+ function joinDocsPath(docsUrl, suffix) {
138
+ return `${docsUrl.replace(/\/+$/, '')}/${suffix.replace(/^\/+/, '')}`;
139
+ }
140
+ /**
141
+ * The single source for the account-sessions guide slug — referenced by
142
+ * BOTH `COMMAND_DOC_PATHS` and `FAILURE_KIND_DOC_PATHS` below. Those two
143
+ * tables key on different fields (a CLI command's full path name vs a run's
144
+ * `failureKind`) and legitimately stay separate, but they name the SAME
145
+ * guide, so the path literal itself must exist exactly once: this is the
146
+ * "same fact computed in two places" defect class this module's own doc
147
+ * comment says it exists to prevent, and it has recurred repeatedly across
148
+ * this epic. Renaming the guide used to require editing two literals in
149
+ * lockstep — miss one and one surface (a run's `docs` field, or `spec
150
+ * --json`'s per-command deep link) silently 404s while the other still
151
+ * works (trawl_cli#185 review).
152
+ */
153
+ const ACCOUNT_SESSIONS_GUIDE_PATH = 'build-your-scrap/account-sessions';
154
+ /**
155
+ * Per-command deep links (issue #185 scope item 1) — a command's FULL path
156
+ * name (matches `CliSpecCommand.name` in spec.ts, e.g.
157
+ * `"scraps account session set"`) is matched against `prefix` either
158
+ * exactly or as a whole path SEGMENT prefix (`startsWith(prefix + ' ')`) —
159
+ * never a bare substring, so a hypothetical future `scraps accounting`
160
+ * command could never false-match the `scraps account` entry below.
161
+ */
162
+ const COMMAND_DOC_PATHS = Object.freeze([
163
+ // Every account/session-management command — set, delete, clear-session,
164
+ // status, and the nested `session set` — is covered by one prefix entry
165
+ // rather than one row per leaf, so a new leaf added under `scraps account`
166
+ // later (e.g. a future `session capture`, trawl_cli#183) inherits the
167
+ // link for free instead of needing its own row.
168
+ { prefix: 'scraps account', path: ACCOUNT_SESSIONS_GUIDE_PATH },
169
+ ]);
170
+ function docsPathForCommand(commandName) {
171
+ const match = COMMAND_DOC_PATHS.find((e) => commandName === e.prefix || commandName.startsWith(`${e.prefix} `));
172
+ return match ? match.path : null;
173
+ }
174
+ /**
175
+ * `resolveCommandDocsUrl` returns `undefined` (never a bare path or a
176
+ * relative link) whenever any of three things is missing — `docsUrl`
177
+ * unresolved (rung 3 already fired), `docsUrl` resolved but NOT derived
178
+ * (rung 1 — a server-declared root; see `DocsUrls.docsUrlIsDerived`'s doc
179
+ * comment for why appending our own guide slug onto a third party's root is
180
+ * exactly the confidently-wrong-404 defect this gate closes), or no guide is
181
+ * mapped for this command — so a spec consumer never has to special-case a
182
+ * partial value. Takes the whole resolved `DocsUrls` object (never a bare
183
+ * string) specifically so this provenance can never again be flattened away
184
+ * before it reaches here.
185
+ */
186
+ export function resolveCommandDocsUrl(commandName, docs) {
187
+ if (!docs.docsUrl || !docs.docsUrlIsDerived)
188
+ return undefined;
189
+ const path = docsPathForCommand(commandName);
190
+ return path ? joinDocsPath(docs.docsUrl, path) : undefined;
191
+ }
192
+ /**
193
+ * failureKind -> guide path (issue #185 scope item 2). Keyed on trawl_node's
194
+ * run-level `failureKind` field (trawl_node#1975, surfaced by this CLI's
195
+ * own `doctor`/`data --errors`/`run-info` JSON payloads since cli#182) —
196
+ * this is a DIFFERENT axis from errors.ts's `ErrorEnvelope.kind` (a CLI
197
+ * transport/execution outcome like `"auth"` meaning "you're not logged
198
+ * into trawl", `"network"`, `"usage"`, …). Both happen to use the string
199
+ * `"auth"` for unrelated things: this table's `'auth'` means "the SCRAPED
200
+ * TARGET site showed a login wall", errors.ts's `'auth'` means "the TRAWL
201
+ * API rejected your OWN credentials". Never merge these two tables — they
202
+ * are keyed off different fields on different objects, and conflating them
203
+ * would either miss the run-level guide or wrongly attach it to an
204
+ * unrelated CLI auth failure.
205
+ */
206
+ const FAILURE_KIND_DOC_PATHS = Object.freeze({
207
+ auth: ACCOUNT_SESSIONS_GUIDE_PATH,
208
+ });
209
+ function docsPathForFailureKind(kind) {
210
+ if (!kind)
211
+ return null;
212
+ return FAILURE_KIND_DOC_PATHS[kind] ?? null;
213
+ }
214
+ /** Same "undefined unless docsUrl is resolved AND derived (rung 2)" gate as
215
+ * `resolveCommandDocsUrl` above, same reason — never append this CLI's own
216
+ * guide slug onto a rung-1 server-declared root. Callers MUST gate the call
217
+ * on their own staleness rule first (e.g. `doctor.ts`'s `isAuthWall`) — this
218
+ * function only knows the string-to-path mapping, not whether a stamped
219
+ * `failureKind` is still live on a run patched afterward. */
220
+ export function resolveFailureKindDocsUrl(kind, docs) {
221
+ if (!docs.docsUrl || !docs.docsUrlIsDerived)
222
+ return undefined;
223
+ const path = docsPathForFailureKind(kind);
224
+ return path ? joinDocsPath(docs.docsUrl, path) : undefined;
225
+ }
226
+ /**
227
+ * `--help` footer text (issue #185 scope item 3) — a single dim line, for
228
+ * humans, e.g. "Docs: https://trawl.me/docs". Returns `undefined` (never an
229
+ * empty or guessed line) when no `docsUrl` resolved — mirrors
230
+ * lib/tips.ts's shape (a pure function separated from IO, so it's testable
231
+ * without mocking chalk/console) but NOT its TTY gate: tips.ts suppresses a
232
+ * promotional nudge from piped output, but a docs line inside `--help |
233
+ * less` is exactly the kind of thing worth keeping. Un-colored here — the
234
+ * caller (index.ts) applies `chalk.dim` so this stays trivially testable.
235
+ */
236
+ export function docsFooterLine(docsUrl) {
237
+ return docsUrl ? `Docs: ${docsUrl}` : undefined;
238
+ }
@@ -0,0 +1,8 @@
1
+ /**
2
+ * @desc Returns a factual reason string when `apiUrl` is not an
3
+ * acceptable transport for a session upload, or null when it is
4
+ * (`https:`, or any scheme against a loopback host). An unparseable URL
5
+ * returns null — it fails with its own clear error at the actual fetch
6
+ * call, not here.
7
+ */
8
+ export declare function assertSecureTransport(apiUrl: string): string | null;
@@ -0,0 +1,39 @@
1
+ /**
2
+ * Refuses a plaintext API base for the two commands that ship a browser
3
+ * session — a bearer-equivalent secret — in the request body
4
+ * (trawl_cli#183 review finding 5). Reproduced: `TRAWL_API_URL=http://…
5
+ * trawl scraps account session capture <id>` uploaded the full session in
6
+ * cleartext with no warning; this command is the first in the CLI to PUT
7
+ * that kind of secret, so the fix is scoped HERE rather than to the
8
+ * generic `request()` in api.ts — every other command's blast radius
9
+ * stays exactly what it was.
10
+ *
11
+ * Loopback hosts are exempt (by exact hostname only — a plaintext host
12
+ * that merely happens to RESOLVE to loopback is not detected here) so
13
+ * local/self-hosted dev and the test suite keep working over plain HTTP.
14
+ */
15
+ // `new URL(...).hostname` returns the IPv6 form WITH its brackets
16
+ // (`'[::1]'`, never bare `'::1'`) — both are listed defensively so this
17
+ // never regresses silently if that ever changes.
18
+ const LOOPBACK_HOSTS = new Set(['127.0.0.1', '::1', '[::1]', 'localhost']);
19
+ /**
20
+ * @desc Returns a factual reason string when `apiUrl` is not an
21
+ * acceptable transport for a session upload, or null when it is
22
+ * (`https:`, or any scheme against a loopback host). An unparseable URL
23
+ * returns null — it fails with its own clear error at the actual fetch
24
+ * call, not here.
25
+ */
26
+ export function assertSecureTransport(apiUrl) {
27
+ let url;
28
+ try {
29
+ url = new URL(apiUrl);
30
+ }
31
+ catch {
32
+ return null;
33
+ }
34
+ if (url.protocol === 'https:')
35
+ return null;
36
+ if (LOOPBACK_HOSTS.has(url.hostname.toLowerCase()))
37
+ return null;
38
+ return `The configured API URL (${url.origin}) is not https: — a captured/uploaded browser session is a bearer-equivalent secret and this command refuses to send it in cleartext. An https:// API URL is accepted (trawl login --url https://…); so is any URL on 127.0.0.1, ::1, or localhost, for local testing.`;
39
+ }
@@ -0,0 +1,21 @@
1
+ /**
2
+ * The non-interactive guard for `scraps account session capture`
3
+ * (trawl_cli#183) — deliberately its OWN function, not a reuse of
4
+ * `confirm.ts`'s `isInteractive`. That one returns false under `--json`,
5
+ * which is wrong here: `--json` + a real interactive terminal is a valid
6
+ * way to run this command (a human still drives the browser; only the
7
+ * final result on stdout is machine-shaped). What actually makes this
8
+ * command impossible is no TTY on stdin (nobody can press Enter) or, on
9
+ * Linux, no display server to open a visible window on.
10
+ */
11
+ export interface NonInteractiveEnv {
12
+ stdinIsTTY: boolean;
13
+ platform: NodeJS.Platform;
14
+ env: NodeJS.ProcessEnv;
15
+ }
16
+ /**
17
+ * @desc Returns a factual reason string when this command cannot work in
18
+ * the current environment, or null when it can proceed. Never guesses —
19
+ * both checks are direct, observable facts (no TTY / no display var).
20
+ */
21
+ export declare function detectNonInteractive({ stdinIsTTY, platform, env }: NonInteractiveEnv): string | null;
@@ -0,0 +1,14 @@
1
+ /**
2
+ * @desc Returns a factual reason string when this command cannot work in
3
+ * the current environment, or null when it can proceed. Never guesses —
4
+ * both checks are direct, observable facts (no TTY / no display var).
5
+ */
6
+ export function detectNonInteractive({ stdinIsTTY, platform, env }) {
7
+ if (!stdinIsTTY) {
8
+ return 'No interactive terminal attached (stdin is not a TTY) — this command opens a visible Chrome window for a human to log into and needs a terminal to confirm completion. It does not work headless, in CI, or piped.';
9
+ }
10
+ if (platform === 'linux' && !env.DISPLAY && !env.WAYLAND_DISPLAY) {
11
+ return 'No display detected (DISPLAY/WAYLAND_DISPLAY are both unset) — this command opens a visible Chrome window and cannot run over a plain SSH session, in a container, or headless.';
12
+ }
13
+ return null;
14
+ }