@trawlme/cli 3.12.0 → 3.12.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +2 -2
- package/dist/commands/create.d.ts +0 -28
- package/dist/commands/create.js +0 -89
- package/dist/commands/doctor.d.ts +0 -79
- package/dist/commands/doctor.js +1 -187
- package/dist/commands/login.js +0 -67
- package/dist/commands/ping.d.ts +0 -15
- package/dist/commands/ping.js +0 -15
- package/dist/commands/scraps.d.ts +0 -120
- package/dist/commands/scraps.js +10 -724
- package/dist/commands/skills.js +0 -22
- package/dist/commands/spec.d.ts +0 -85
- package/dist/commands/spec.js +0 -67
- package/dist/commands/telemetry.js +0 -4
- package/dist/commands/token.js +0 -28
- package/dist/commands/upgrade.js +0 -22
- package/dist/commands/whoami.d.ts +0 -12
- package/dist/commands/whoami.js +0 -6
- package/dist/index.d.ts +0 -188
- package/dist/index.js +0 -349
- package/dist/lib/api.d.ts +0 -78
- package/dist/lib/api.js +1 -320
- package/dist/lib/cdp-pipe.d.ts +0 -72
- package/dist/lib/cdp-pipe.js +1 -81
- package/dist/lib/chrome-discovery.d.ts +0 -11
- package/dist/lib/chrome-discovery.js +0 -19
- package/dist/lib/chrome-launch.d.ts +0 -40
- package/dist/lib/chrome-launch.js +0 -69
- package/dist/lib/config.d.ts +0 -53
- package/dist/lib/config.js +0 -55
- package/dist/lib/confirm.d.ts +0 -55
- package/dist/lib/confirm.js +0 -47
- package/dist/lib/docs.d.ts +0 -123
- package/dist/lib/docs.js +0 -169
- package/dist/lib/errors.d.ts +0 -134
- package/dist/lib/errors.js +0 -151
- package/dist/lib/format.d.ts +0 -6
- package/dist/lib/format.js +0 -6
- package/dist/lib/json.d.ts +0 -35
- package/dist/lib/json.js +0 -48
- package/dist/lib/jwt.d.ts +0 -7
- package/dist/lib/jwt.js +0 -7
- package/dist/lib/pinch.d.ts +0 -53
- package/dist/lib/pinch.js +6 -112
- package/dist/lib/pinchAnimation.d.ts +0 -16
- package/dist/lib/pinchAnimation.js +8 -29
- package/dist/lib/posthog.d.ts +0 -9
- package/dist/lib/posthog.js +0 -23
- package/dist/lib/prompt.js +1 -20
- package/dist/lib/secure-transport.d.ts +0 -7
- package/dist/lib/secure-transport.js +0 -24
- package/dist/lib/session-capture-guard.d.ts +0 -15
- package/dist/lib/session-capture-guard.js +0 -5
- package/dist/lib/session-capture.d.ts +0 -125
- package/dist/lib/session-capture.js +0 -281
- package/dist/lib/skills.d.ts +0 -175
- package/dist/lib/skills.js +1 -216
- package/dist/lib/skillsNudge.d.ts +0 -17
- package/dist/lib/skillsNudge.js +0 -83
- package/dist/lib/spinner.d.ts +0 -39
- package/dist/lib/spinner.js +0 -40
- package/dist/lib/storage-state.d.ts +0 -112
- package/dist/lib/storage-state.js +0 -131
- package/dist/lib/tips.d.ts +0 -38
- package/dist/lib/tips.js +0 -77
- package/dist/lib/updateCheckWorker.js +0 -14
- package/dist/lib/updateNotifier.d.ts +0 -17
- package/dist/lib/updateNotifier.js +0 -53
- package/dist/lib/validate.d.ts +0 -8
- package/dist/lib/validate.js +0 -8
- package/dist/lib/version.d.ts +0 -12
- package/dist/lib/version.js +1 -13
- package/docs/agent-quickstart.md +2 -2
- package/package.json +2 -2
package/dist/commands/doctor.js
CHANGED
|
@@ -1,27 +1,6 @@
|
|
|
1
1
|
import { api } from '../lib/api.js';
|
|
2
2
|
import chalk from 'chalk';
|
|
3
3
|
import { formatDate } from '../lib/format.js';
|
|
4
|
-
/**
|
|
5
|
-
* Known anti-bot vendors that the worker's `block.kind` field may name
|
|
6
|
-
* (ex-flat `blockType`, trawl_node#1950). Ordered by first-match;
|
|
7
|
-
* `block.kind` is a freeform worker string, not an enum (see the `Run.block`
|
|
8
|
-
* doc comment), so this is a best-effort substring match against the real
|
|
9
|
-
* field — never an invented/mocked value.
|
|
10
|
-
*
|
|
11
|
-
* `datadome`/`perimeterx`/`akamai`/`cloudflare` are literal substrings the
|
|
12
|
-
* current worker emits (detect.js). `kasada` is forward-compatible: the issue
|
|
13
|
-
* (#62 / wall-registry WS-7) names it as a genuine wall, but the current
|
|
14
|
-
* worker vocabulary does not yet emit it — this branch lights up
|
|
15
|
-
* automatically if/when a future worker classifier does, without a CLI
|
|
16
|
-
* change.
|
|
17
|
-
*
|
|
18
|
-
* trawl_cli#182 — an `auth` entry used to live here, rendering a login wall
|
|
19
|
-
* as "walled: auth — no reliable bypass". Removed: unlike a real anti-bot
|
|
20
|
-
* vendor, a login wall's own reliable bypass is the user's session cookies,
|
|
21
|
-
* so that verdict was wrong, not just early. `failureKind==='auth'`
|
|
22
|
-
* (trawl_node#1975) is the honest signal for a login wall — see the
|
|
23
|
-
* dedicated branch in `formatDoctor` below.
|
|
24
|
-
*/
|
|
25
4
|
const WALL_VENDOR_PATTERNS = [
|
|
26
5
|
[/datadome/i, 'DataDome'],
|
|
27
6
|
[/kasada/i, 'Kasada'],
|
|
@@ -29,66 +8,18 @@ const WALL_VENDOR_PATTERNS = [
|
|
|
29
8
|
[/akamai/i, 'Akamai'],
|
|
30
9
|
[/cloudflare/i, 'Cloudflare'],
|
|
31
10
|
];
|
|
32
|
-
/**
|
|
33
|
-
* Resolve a known anti-bot vendor name from a worker `block.kind`
|
|
34
|
-
* string. Returns null when the run succeeded, when there is no `block.kind`
|
|
35
|
-
* signal at all, or when `block.kind` names something other than a known
|
|
36
|
-
* vendor (e.g. `proxy-domain-gate`, `rate_limited_per_host`) — those stay on
|
|
37
|
-
* the genuine-error path since we can't honestly attribute them to a specific
|
|
38
|
-
* "no reliable bypass" wall.
|
|
39
|
-
*
|
|
40
|
-
* `blocked` is deliberately NOT consulted here (#177): it is a mid-run,
|
|
41
|
-
* attempt-level signal the worker only stamps on one envelope shape, so it
|
|
42
|
-
* reads `false` on the great majority of genuinely walled runs (the early-block
|
|
43
|
-
* throw path sets `block.kind` but never `blocked` — 63 of 64 walled runs
|
|
44
|
-
* measured on prod). `status === true` wins instead: a run that ultimately
|
|
45
|
-
* returned data was not walled, even if an earlier tier's `block.kind` stamp
|
|
46
|
-
* survived on the row.
|
|
47
|
-
*
|
|
48
|
-
* trawl_node#1950 renamed the flat `blockType` field to nested `block.kind`.
|
|
49
|
-
* Read `block?.kind` first, falling back to the deprecated flat `blockType` —
|
|
50
|
-
* this CLI can be published (and talk to prod) before the trawl_node prod tag
|
|
51
|
-
* containing #1950 is cut (see the `Run.blockType` doc comment), so a prod
|
|
52
|
-
* response can still be the pre-#1950 flat shape for a while.
|
|
53
|
-
*/
|
|
54
11
|
export function detectWallVendor(run) {
|
|
55
|
-
// A run that returned data was not walled. `status === true` is not the whole
|
|
56
|
-
// predicate: trawl_node flips a degraded-but-non-empty run to
|
|
57
|
-
// `status:false, statusDetail:'regression'` (scraps.service.js, #1112) via a
|
|
58
|
-
// patch that never clears `block.kind` — so keying on `status` alone would
|
|
59
|
-
// print "no reliable bypass" next to "Regression: length N vs baseline M",
|
|
60
|
-
// which is the same contradiction this guard exists to remove.
|
|
61
12
|
if (run.status === true || run.statusDetail === 'regression')
|
|
62
13
|
return null;
|
|
63
14
|
const kind = run.block?.kind ?? run.blockType ?? null;
|
|
64
15
|
if (!kind)
|
|
65
|
-
return null;
|
|
16
|
+
return null;
|
|
66
17
|
for (const [pattern, label] of WALL_VENDOR_PATTERNS) {
|
|
67
18
|
if (pattern.test(kind))
|
|
68
19
|
return label;
|
|
69
20
|
}
|
|
70
21
|
return null;
|
|
71
22
|
}
|
|
72
|
-
/**
|
|
73
|
-
* #184 — true for a run currently carrying a live (non-stale) login-wall
|
|
74
|
-
* verdict. Extracted out of `formatDoctor`'s own local `authWall` const so
|
|
75
|
-
* `commands/scraps.ts`'s `doctor`/`run-info` actions can reuse the EXACT
|
|
76
|
-
* same guard to decide whether to fire the skills-install safety-net nudge
|
|
77
|
-
* (lib/skillsNudge.ts `maybeSuggestSkillsForAuthWall`) — one rule, not two
|
|
78
|
-
* copies that could quietly drift apart.
|
|
79
|
-
*
|
|
80
|
-
* Same staleness guard as `detectWallVendor` above and for the same reason:
|
|
81
|
-
* `failureKind` is a terminal classification trawl_node stamps once, but
|
|
82
|
-
* `patchForRegression` can flip `status`/`statusDetail` to success/
|
|
83
|
-
* regression LATER without ever clearing it — so a run that ultimately
|
|
84
|
-
* succeeded or degraded must never still read as an active auth wall.
|
|
85
|
-
*
|
|
86
|
-
* Typed structurally loose (not `Pick<Run, ...>`) on purpose: `Run.status`/
|
|
87
|
-
* `statusDetail` are required fields, but `scraps.ts`'s own `HistoryRun`
|
|
88
|
-
* (the run-info command's shape) declares the same three fields OPTIONAL —
|
|
89
|
-
* a `Pick<Run, ...>` parameter type would reject that caller at compile
|
|
90
|
-
* time even though every field it actually reads is present at runtime.
|
|
91
|
-
*/
|
|
92
23
|
export function isAuthWall(run) {
|
|
93
24
|
return run.failureKind === 'auth' && run.status !== true && run.statusDetail !== 'regression';
|
|
94
25
|
}
|
|
@@ -120,28 +51,8 @@ export function pickFix(fix) {
|
|
|
120
51
|
.filter((k) => k in fix)
|
|
121
52
|
.map((k) => [k, fix[k]]));
|
|
122
53
|
}
|
|
123
|
-
/**
|
|
124
|
-
* Pure formatter — returns a human-readable diagnosis string for a run.
|
|
125
|
-
* Used by `doctor`, `doctor --autofix`, and `data --errors`.
|
|
126
|
-
*
|
|
127
|
-
* @param scrapTitle - Display name of the scrap
|
|
128
|
-
* @param run - Owner-safe history run object
|
|
129
|
-
* @param fix - Optional autofix activity (null = no fix attempted)
|
|
130
|
-
* @param scrapId - Scrap document id (used in hint lines — autofix/snapshot take scrap id, not run id)
|
|
131
|
-
* @returns Multi-line string ready for console.log
|
|
132
|
-
*/
|
|
133
54
|
export function formatDoctor(scrapTitle, run, fix = null, scrapId) {
|
|
134
55
|
const lines = [];
|
|
135
|
-
// Header + status badge. status:null means a run is IN FLIGHT (node
|
|
136
|
-
// persists {status:null, statusDetail:null, inFlight:true} the moment a
|
|
137
|
-
// run starts) — that must never render as "failed". (#88 item 1)
|
|
138
|
-
//
|
|
139
|
-
// #91 LOW — a regression row (status:false, statusDetail:'regression') was
|
|
140
|
-
// falling into the same red "failed" bucket as a genuine failure, then
|
|
141
|
-
// showing a contradictory "Regression: X vs Y" detail line right below it.
|
|
142
|
-
// A regression run's write actually SUCCEEDED (the item count just dropped
|
|
143
|
-
// vs baseline, detected by an async patch afterward — see #88 item 2 in
|
|
144
|
-
// scraps.ts) — it needs its own distinct badge, not "failed".
|
|
145
56
|
const ok = run.status === true;
|
|
146
57
|
const badge = ok
|
|
147
58
|
? chalk.green('● success')
|
|
@@ -154,69 +65,8 @@ export function formatDoctor(scrapTitle, run, fix = null, scrapId) {
|
|
|
154
65
|
: chalk.red('● failed');
|
|
155
66
|
lines.push(`${chalk.bold(scrapTitle)} ${badge}${run.statusDetail ? ` (${run.statusDetail})` : ''}`);
|
|
156
67
|
lines.push(chalk.dim(` Run ID: ${run._id}`));
|
|
157
|
-
// Error message — an honest accept-wall string for known-walled scraps
|
|
158
|
-
// (DataDome/Kasada/PerimeterX/Akamai terminal verdict from the worker),
|
|
159
|
-
// otherwise the real error (genuine transient failure).
|
|
160
|
-
//
|
|
161
|
-
// trawl_cli#182 — the wall-vendor verdict (this CLI's own
|
|
162
|
-
// WALL_VENDOR_PATTERNS matched against block.kind/blockType) and the
|
|
163
|
-
// login-wall hint (node's failureKind==='auth', trawl_node#1975) are two
|
|
164
|
-
// INDEPENDENT server-derived signals with no shared contract: node's block
|
|
165
|
-
// classifier (modules/historys/helpers/failureKind.js BLOCK_TYPE_PATTERN)
|
|
166
|
-
// and this CLI's vendor table live in different repos and can legitimately
|
|
167
|
-
// disagree. Concretely, `kasada` matches this CLI's table but is absent
|
|
168
|
-
// from node's block pattern, and the deprecated flat `blockType` fallback
|
|
169
|
-
// is a field node's classifier no longer reads at all — so a run can come
|
|
170
|
-
// back `failureKind:'auth'` (the server says login wall) while also
|
|
171
|
-
// matching a vendor here (this CLI says Kasada/DataDome/etc), both true
|
|
172
|
-
// signals about the same run, neither one wrong.
|
|
173
|
-
//
|
|
174
|
-
// trawl_cli#169 precedent: never assert client-side what only the server
|
|
175
|
-
// knows. Silently picking a winner between two disagreeing signals is
|
|
176
|
-
// exactly that — printing "no reliable bypass" alone asserts the run is
|
|
177
|
-
// NOT an auth wall (may stop someone from trying a fixable cookie
|
|
178
|
-
// problem); printing the cookie hint alone asserts the run is NOT a
|
|
179
|
-
// vendor wall (may send someone chasing cookies against a wall no cookie
|
|
180
|
-
// fixes). This CLI cannot tell which is true without either importing
|
|
181
|
-
// node's BLOCK_TYPE_PATTERN (a second source of truth for the same list —
|
|
182
|
-
// how this defect got here) or re-detecting the login URL itself (the
|
|
183
|
-
// client-side re-derivation this fix deliberately rejects). So when both
|
|
184
|
-
// fire, render ONE hedged line naming both readings instead of picking:
|
|
185
|
-
// never emit the bare "no reliable bypass" verdict in that case — it is
|
|
186
|
-
// exactly the false precision this branch exists to avoid.
|
|
187
|
-
//
|
|
188
|
-
// `conflictingSignals` is deliberately NOT gated on `scrapId` the way the
|
|
189
|
-
// plain hint below is: the false "no reliable bypass" verdict is wrong
|
|
190
|
-
// regardless of whether we also have an id to build an actionable command
|
|
191
|
-
// for, so a scrapId-less conflicting run must still avoid it. Only the
|
|
192
|
-
// action line degrades (to a generic Settings pointer) when there's no id.
|
|
193
|
-
//
|
|
194
|
-
// Single-signal cases are untouched: a plain vendor match (no
|
|
195
|
-
// failureKind:'auth') still gets "no reliable bypass"; a plain
|
|
196
|
-
// failureKind:'auth' (no vendor match) still gets the plain hint block
|
|
197
|
-
// below. `errMsg` stays suppressed whenever any hint form (conflicting or
|
|
198
|
-
// plain) is about to render — node's own `errorMessage` for an auth run
|
|
199
|
-
// (historys.service.js ~line 774) IS the hint copy verbatim, literal
|
|
200
|
-
// `<id>` placeholder and all, so printing it here would just be a second
|
|
201
|
-
// rendering of the same information. If neither hint form renders (no
|
|
202
|
-
// vendor conflict and no scrapId for the plain hint), fall back to node's
|
|
203
|
-
// raw message rather than dropping the error entirely.
|
|
204
68
|
const wallVendor = detectWallVendor(run);
|
|
205
69
|
const errMsg = run.errorMessage ?? run.errorSnapshot?.errorMessage;
|
|
206
|
-
// trawl_cli#182 — same `status`/`statusDetail` guard as `detectWallVendor`
|
|
207
|
-
// above, and for the same reason (see its doc comment): `failureKind` is a
|
|
208
|
-
// terminal classification trawl_node stamps once
|
|
209
|
-
// (modules/historys/services/historys.service.js ~line 837), but
|
|
210
|
-
// `patchForRegression` (modules/scraps/services/scraps.service.js ~line
|
|
211
|
-
// 1972) can flip `status`/`statusDetail` to success/regression LATER,
|
|
212
|
-
// without ever clearing `failureKind`. Left unguarded, a run that
|
|
213
|
-
// ultimately succeeded or degraded could still carry a stale
|
|
214
|
-
// `failureKind:'auth'` and render the login-wall hint (or the
|
|
215
|
-
// conflicting-signals hedge) right next to a green "success" badge or a
|
|
216
|
-
// "Regression: length N vs baseline M" line — the exact contradiction
|
|
217
|
-
// `detectWallVendor`'s guard already exists to prevent for `block.kind`.
|
|
218
|
-
// One rule, not two coincidences: both checks guard the same two fields
|
|
219
|
-
// against the same after-the-fact patch.
|
|
220
70
|
const authWall = isAuthWall(run);
|
|
221
71
|
const authHintWillRender = authWall && Boolean(scrapId);
|
|
222
72
|
const conflictingSignals = Boolean(wallVendor) && authWall;
|
|
@@ -232,69 +82,38 @@ export function formatDoctor(scrapTitle, run, fix = null, scrapId) {
|
|
|
232
82
|
else if (errMsg && !authHintWillRender) {
|
|
233
83
|
lines.push(chalk.dim(' Error: ') + chalk.red(errMsg));
|
|
234
84
|
}
|
|
235
|
-
// Failed selector
|
|
236
85
|
if (run.errorSnapshot?.selector) {
|
|
237
86
|
lines.push(chalk.dim(' Failed selector: ') + chalk.cyan(run.errorSnapshot.selector));
|
|
238
87
|
}
|
|
239
|
-
// Blocked
|
|
240
88
|
lines.push(chalk.dim(' Blocked: ') + (run.blocked ? chalk.red('yes') : 'no'));
|
|
241
|
-
// Proxy tier (abstract, kept for owner)
|
|
242
89
|
if (run.proxyTier && TIER_LABELS[run.proxyTier]) {
|
|
243
90
|
lines.push(chalk.dim(' Proxy: ') + TIER_LABELS[run.proxyTier]);
|
|
244
91
|
}
|
|
245
|
-
// Empty context detail
|
|
246
92
|
if (run.statusDetail === 'empty' && run.emptyContext?.page) {
|
|
247
93
|
const page = run.emptyContext.page;
|
|
248
94
|
lines.push(chalk.dim(' Empty context: ') + `url=${page.url ?? '?'} anchors=${page.totalAnchors ?? '?'}`);
|
|
249
95
|
}
|
|
250
|
-
// Login-wall hint (trawl_node#1975 `failureKind==='auth'`) — trust node's
|
|
251
|
-
// verdict, never re-detect the login URL client-side. Hypothesis-framed
|
|
252
|
-
// ("likely fix"), not a promise: measured cases (Reddit/X/Instagram) had
|
|
253
|
-
// valid cookies and stayed walled. Gated on `scrapId` alone (via
|
|
254
|
-
// `authHintWillRender` above, shared with the generic error line so the
|
|
255
|
-
// two can't drift apart), NOT the `scrapId ?? run._id` fallback used
|
|
256
|
-
// elsewhere here — `run._id` is a history id, and the session-set command
|
|
257
|
-
// on one silently targets the wrong document.
|
|
258
|
-
//
|
|
259
|
-
// trawl_cli#182 — excludes `conflictingSignals`: when a wall-vendor match
|
|
260
|
-
// also fired, the combined hedged line above already covers the cookie
|
|
261
|
-
// suggestion. Rendering this plain, unhedged block too would stack a
|
|
262
|
-
// second verdict under the first (and directly contradict it, since this
|
|
263
|
-
// block asserts the run IS a login wall with no caveat) — exactly the
|
|
264
|
-
// two-verdicts-for-one-run defect this fix removes. Exactly one verdict
|
|
265
|
-
// block renders per run.
|
|
266
96
|
if (authHintWillRender && !conflictingSignals) {
|
|
267
97
|
lines.push('');
|
|
268
98
|
lines.push(chalk.yellow(' Login wall (hypothesis): ') + 'this looks like a login redirect — your own session cookies are the likely fix.');
|
|
269
99
|
lines.push(chalk.dim(` → trawl scraps account session set ${scrapId} -c cookies.json (or app Settings → Account)`));
|
|
270
100
|
}
|
|
271
|
-
// Regression detail. trawl_node#1950 dropped the redundant `regressionDetected`
|
|
272
|
-
// boolean — it was `true` on every row iff `statusDetail === 'regression'`,
|
|
273
|
-
// so read that directly (works against a pre- or post-#1950 backend alike,
|
|
274
|
-
// no CLI/backend sequencing gap: `statusDetail` isn't part of this rename).
|
|
275
101
|
if (run.statusDetail === 'regression') {
|
|
276
102
|
lines.push(chalk.dim(' Regression: ') + `length ${run.length ?? '?'} vs baseline ${run.baselineLength ?? '?'}`);
|
|
277
103
|
}
|
|
278
|
-
// Autofix summary (when fix exists)
|
|
279
104
|
if (fix) {
|
|
280
105
|
lines.push(` ${chalk.magenta('Autofix:')} ${fix.outcome ?? '?'}${fix.classification ? ` — ${fix.classification}` : ''}${fix.reason ? ` (${fix.reason})` : ''}`);
|
|
281
106
|
lines.push(chalk.dim(` → full diff/dry-run/knowledge: trawl scraps autofix ${scrapId ?? run._id} (or doctor --autofix)`));
|
|
282
107
|
}
|
|
283
|
-
// Timestamp
|
|
284
108
|
if (run.createdAt) {
|
|
285
109
|
lines.push(chalk.dim(' Run at: ') + formatDate(run.createdAt));
|
|
286
110
|
}
|
|
287
|
-
// Snapshot hint
|
|
288
111
|
if (run.errorSnapshot?.html || run.statusDetail === 'empty' || run.statusDetail === 'error') {
|
|
289
112
|
lines.push('');
|
|
290
113
|
lines.push(chalk.dim(` → trawl scraps snapshot ${scrapId ?? run._id} --error (download error-path HTML)`));
|
|
291
114
|
}
|
|
292
115
|
return lines.join('\n');
|
|
293
116
|
}
|
|
294
|
-
/**
|
|
295
|
-
* Pure formatter — returns a human-readable autofix detail block.
|
|
296
|
-
* Shows diff, dry-run results, knowledge consulted.
|
|
297
|
-
*/
|
|
298
117
|
export function formatAutofix(fix) {
|
|
299
118
|
const lines = [];
|
|
300
119
|
lines.push(chalk.magenta.bold('Autofix attempt'));
|
|
@@ -324,17 +143,12 @@ export function formatAutofix(fix) {
|
|
|
324
143
|
}
|
|
325
144
|
return lines.join('\n');
|
|
326
145
|
}
|
|
327
|
-
/**
|
|
328
|
-
* Fetch the latest run + its ai_fix_end activity for a scrap.
|
|
329
|
-
* Returns null if the scrap has no runs yet.
|
|
330
|
-
*/
|
|
331
146
|
export async function fetchRunAndFix(scrapId) {
|
|
332
147
|
const scrap = await api.get(`/api/scraps/${scrapId}`);
|
|
333
148
|
const hid = scrap.history?.[0]?._id;
|
|
334
149
|
if (!hid)
|
|
335
150
|
return null;
|
|
336
151
|
const run = await api.get(`/api/historys/${hid}`);
|
|
337
|
-
// api.get unwraps the { data: T } envelope — the response is the array directly
|
|
338
152
|
const acts = await api.get(`/api/scraps/${scrapId}/activities?history=${hid}&limit=5`);
|
|
339
153
|
const fixAct = Array.isArray(acts) ? acts.find((a) => a.type === 'ai_fix_end') : undefined;
|
|
340
154
|
const fix = fixAct ? { ...fixAct.metadata, createdAt: fixAct.createdAt } : null;
|
package/dist/commands/login.js
CHANGED
|
@@ -28,60 +28,11 @@ async function promptEmail() {
|
|
|
28
28
|
rl.close();
|
|
29
29
|
}
|
|
30
30
|
}
|
|
31
|
-
/** #107 — the shared success-reporting tail for all three login paths (env
|
|
32
|
-
* token, --token flag, email/password). Under --json, prints a single
|
|
33
|
-
* structured result on stdout instead of the chalk lines — never the raw
|
|
34
|
-
* token itself (that's what `trawl token` is for). */
|
|
35
31
|
function reportLoginSuccess(opts, email) {
|
|
36
|
-
// #169 review finding 3 — stderr, both under --json and plain output: this
|
|
37
|
-
// is a warning about what happens AFTER login succeeds, not part of either
|
|
38
|
-
// payload shape, so it must never be silently dropped just because --json
|
|
39
|
-
// is set.
|
|
40
32
|
const overrideVar = getLiveAuthEnvVar();
|
|
41
33
|
if (overrideVar) {
|
|
42
34
|
console.error(chalk.yellow(`⚠ ${overrideVar} is set in your environment — it overrides the token just stored here for every subsequent command until you unset it.`));
|
|
43
35
|
}
|
|
44
|
-
// #184 — `login` is the ONE intentional human setup moment: bootstrap any
|
|
45
|
-
// bundled Claude skill the user doesn't have yet. Gated exactly like
|
|
46
|
-
// every other filesystem-mutating side effect in this CLI
|
|
47
|
-
// (json/TTY/opt-out — see lib/skills.ts's isSkillsActionAllowed), so a
|
|
48
|
-
// CI/agent `TRAWL_TOKEN=x trawl login --json` never writes into
|
|
49
|
-
// ~/.claude/skills on a machine with no Claude Code session to discover
|
|
50
|
-
// them. `isInteractive` is the exact TTY definition login already uses
|
|
51
|
-
// for its own email/password prompt gate (lib/confirm.ts) — stdin AND
|
|
52
|
-
// stdout, not just stdout. `bootstrapped` is `null` ONLY when gated out
|
|
53
|
-
// (json/non-TTY/opted-out) or when every bundled skill is already owned
|
|
54
|
-
// by trawl — the genuine "nothing to do" case, where the lines below
|
|
55
|
-
// naturally never print. It is non-null whenever anything was attempted,
|
|
56
|
-
// whether or not any of it actually landed — see the block below.
|
|
57
|
-
//
|
|
58
|
-
// stderr, same convention as the `overrideVar` warning right above (and
|
|
59
|
-
// for the same reason): this is orthogonal to the `--json` envelope's
|
|
60
|
-
// fixed shape below, so it must never risk landing on stdout — a defense
|
|
61
|
-
// that holds even if `bootstrapSkillsOnLogin`'s own json/TTY gate above it
|
|
62
|
-
// were ever wrong, not just a style match.
|
|
63
|
-
//
|
|
64
|
-
// #184 review (BLOCK + MAJOR) — `bootstrapped` is non-null whenever there
|
|
65
|
-
// was anything to attempt, whether or not any of it actually landed, so
|
|
66
|
-
// both halves below must be checked independently: `installed` prints the
|
|
67
|
-
// success line, `skipped` prints one honest line per skill that was
|
|
68
|
-
// requested but did NOT land (a permission error, or a pre-existing
|
|
69
|
-
// marker-less dir the ownership guard refused to overwrite) and WHY —
|
|
70
|
-
// `reason` is already a relayable, fact-only string (SkillOwnershipRefusalError's
|
|
71
|
-
// `.relayableReason`, or a raw fs error message) with no imperative, so it
|
|
72
|
-
// is safe to print verbatim on this channel. A total failure (installed
|
|
73
|
-
// empty, skipped non-empty) must read as "attempted and failed" — never
|
|
74
|
-
// fall through to silence, which is indistinguishable from "never
|
|
75
|
-
// attempted" (the exact false-success shape this issue exists to
|
|
76
|
-
// prevent). The restart note only applies to what actually landed, so it
|
|
77
|
-
// stays scoped to that branch.
|
|
78
|
-
//
|
|
79
|
-
// #184 defect 2 — `error` is the THIRD case: distinct from both "nothing
|
|
80
|
-
// to do" (bootstrapped is `null`, nothing prints) and "some/all skills
|
|
81
|
-
// failed" (`skipped`, above) — it means bootstrapSkillsOnLogin could not
|
|
82
|
-
// even determine which skills to install (the bundled skills package
|
|
83
|
-
// looks missing/corrupted). Stated as a fact, same convention as every
|
|
84
|
-
// other line here.
|
|
85
36
|
const bootstrapped = bootstrapSkillsOnLogin({ json: opts.json, isTTY: isInteractive(opts) });
|
|
86
37
|
if (bootstrapped) {
|
|
87
38
|
if (bootstrapped.error) {
|
|
@@ -113,17 +64,11 @@ export const login = new Command('login')
|
|
|
113
64
|
.option('-p, --password <password>', 'Password (CI only — visible in process list and shell history)')
|
|
114
65
|
.option('--json', 'Output as JSON')
|
|
115
66
|
.action(async (opts) => {
|
|
116
|
-
// Validate --url but do NOT persist it yet. A failed signin must not
|
|
117
|
-
// brick the config by pointing it at an unreachable/wrong host while the
|
|
118
|
-
// OLD token stays stored (and would then be sent to that new host on the
|
|
119
|
-
// next command). Target this run via `pendingUrl` and persist only once
|
|
120
|
-
// a token has actually been obtained. (#68)
|
|
121
67
|
const pendingUrl = opts.url ? requireUrl(opts.url, '--url') : undefined;
|
|
122
68
|
const persistUrlIfPending = () => {
|
|
123
69
|
if (pendingUrl)
|
|
124
70
|
config.set('apiUrl', pendingUrl);
|
|
125
71
|
};
|
|
126
|
-
// Check environment variable override first
|
|
127
72
|
const envToken = process.env['TRAWL_TOKEN'];
|
|
128
73
|
if (envToken) {
|
|
129
74
|
config.set('token', requireFreshJwt(envToken, 'TRAWL_TOKEN'));
|
|
@@ -131,20 +76,15 @@ export const login = new Command('login')
|
|
|
131
76
|
reportLoginSuccess(opts);
|
|
132
77
|
return;
|
|
133
78
|
}
|
|
134
|
-
// --token flag: direct JWT (CI / retrocompat)
|
|
135
79
|
if (opts.token) {
|
|
136
80
|
config.set('token', requireFreshJwt(opts.token, '--token'));
|
|
137
81
|
persistUrlIfPending();
|
|
138
82
|
reportLoginSuccess(opts);
|
|
139
83
|
return;
|
|
140
84
|
}
|
|
141
|
-
// Email/password flow
|
|
142
85
|
if (opts.password) {
|
|
143
86
|
console.error(chalk.yellow('Warning: passing --password on the command line is insecure.'));
|
|
144
87
|
}
|
|
145
|
-
// #107 — never blocks on a readline prompt under --json or a non-TTY
|
|
146
|
-
// invocation (agent/CI subprocess); refuses with a clear, structured
|
|
147
|
-
// usage error instead, naming the flag (or TRAWL_TOKEN) to pass.
|
|
148
88
|
if (!opts.email) {
|
|
149
89
|
requireInteractive('Email is required — pass -e/--email, --token, or set TRAWL_TOKEN (refusing to block on a prompt, non-interactive).', { json: opts.json });
|
|
150
90
|
}
|
|
@@ -154,7 +94,6 @@ export const login = new Command('login')
|
|
|
154
94
|
const email = opts.email ?? (await promptEmail());
|
|
155
95
|
const password = opts.password ?? (await promptPassword('Password: '));
|
|
156
96
|
const { data, headers } = await api.publicPost('/api/auth/signin', { email, password }, pendingUrl);
|
|
157
|
-
// Try token from response body first, then fall back to Set-Cookie header
|
|
158
97
|
let raw = typeof data === 'string' ? data : data.token;
|
|
159
98
|
if (!raw) {
|
|
160
99
|
const setCookie = headers.get('set-cookie') ?? '';
|
|
@@ -175,12 +114,6 @@ export const logout = new Command('logout')
|
|
|
175
114
|
.option('--json', 'Output as JSON')
|
|
176
115
|
.action((opts) => {
|
|
177
116
|
config.set('token', '');
|
|
178
|
-
// #169 review finding 3 — this is the sharpest version of the gap: an
|
|
179
|
-
// operator runs `logout` SPECIFICALLY to kill access, and if
|
|
180
|
-
// TRAWL_API_KEY/TRAWL_TOKEN is set, access is NOT killed — every
|
|
181
|
-
// subsequent command keeps authenticating with the env credential this
|
|
182
|
-
// command cannot touch. The wording below must not read as a variant of
|
|
183
|
-
// "logged out"; it says plainly that access is still live.
|
|
184
117
|
const overrideVar = getLiveAuthEnvVar();
|
|
185
118
|
if (overrideVar) {
|
|
186
119
|
console.error(chalk.yellow(`⚠ ${overrideVar} is still set in your environment — access is NOT revoked. Every subsequent command will keep authenticating with it until you unset it (or revoke the key in the dashboard).`));
|
package/dist/commands/ping.d.ts
CHANGED
|
@@ -1,19 +1,4 @@
|
|
|
1
1
|
import { Command } from 'commander';
|
|
2
|
-
/**
|
|
3
|
-
* `GET /api/health` response shape (home.controller.js#health,
|
|
4
|
-
* home.service.js#getHealthStatus) — MCP `trawl_health_ping` parity
|
|
5
|
-
* (`{ok, version, uptime}`). The REST route enriches the payload with
|
|
6
|
-
* `version`/`uptime`/`db`/`memory` ONLY for an admin caller
|
|
7
|
-
* (home.controller.js's `isAdmin` gate); a non-admin JWT gets `{status:'ok'}`
|
|
8
|
-
* alone on a healthy server. A non-2xx response (degraded, `status:503`)
|
|
9
|
-
* throws an ApiError like any other failed request — but a 2xx response is
|
|
10
|
-
* NOT a guarantee that `status === 'ok'` (#106 review F5): a soft-degraded
|
|
11
|
-
* signal (e.g. a partial dependency outage) could still come back as a 200
|
|
12
|
-
* with a non-"ok" `status`. The HTTP call not throwing only means the
|
|
13
|
-
* transport succeeded, never that the reported health is good — so the
|
|
14
|
-
* human-mode `✓ OK` line is gated on the actual field, not assumed from a
|
|
15
|
-
* successful round-trip.
|
|
16
|
-
*/
|
|
17
2
|
export interface PingResponse {
|
|
18
3
|
status: string;
|
|
19
4
|
db?: string;
|
package/dist/commands/ping.js
CHANGED
|
@@ -6,33 +6,18 @@ export const ping = new Command('ping')
|
|
|
6
6
|
.description('Health/version handshake against the Trawl API')
|
|
7
7
|
.option('--json', 'Output as JSON')
|
|
8
8
|
.action(async (opts) => {
|
|
9
|
-
// #148 — the server route is optionalAuth (public, enriched for an admin
|
|
10
|
-
// JWT — see the PingResponse doc above); `api.publicGet` mirrors that:
|
|
11
|
-
// it attaches a stored/env token when one happens to be available but
|
|
12
|
-
// never REQUIRES one, so `ping` works as a zero-config sanity check even
|
|
13
|
-
// with no `trawl login` ever run. `api.get` (the authenticated path)
|
|
14
|
-
// would throw notLoggedInError() locally before this ever reached the
|
|
15
|
-
// network, defeating the whole point of a pre-login handshake.
|
|
16
9
|
const data = await api.publicGet('/api/health');
|
|
17
|
-
// #106 review (kimi) — honest exit code: a soft-degraded 200 (status !== 'ok')
|
|
18
|
-
// must exit non-zero so `trawl ping || handle_degraded` doesn't treat a
|
|
19
|
-
// degraded API as healthy. Applies in BOTH --json and human modes, alongside
|
|
20
|
-
// the raw/decorated output below.
|
|
21
10
|
if (data.status !== 'ok')
|
|
22
11
|
process.exitCode = 1;
|
|
23
12
|
if (opts.json) {
|
|
24
13
|
json(data);
|
|
25
14
|
return;
|
|
26
15
|
}
|
|
27
|
-
// #119 — label it as the API's version, not a bare `v0.4.0` that reads as
|
|
28
|
-
// "the platform is v0.4" (it's the server package version field).
|
|
29
16
|
const versionSuffix = data.version ? ` (api v${data.version})` : '';
|
|
30
17
|
if (data.status === 'ok') {
|
|
31
18
|
console.log(chalk.green('✓ OK') + versionSuffix);
|
|
32
19
|
}
|
|
33
20
|
else {
|
|
34
|
-
// #106 review F5 — never green-light a degraded API just because the
|
|
35
|
-
// HTTP call didn't throw; show the real reported status instead.
|
|
36
21
|
console.log(chalk.yellow(`⚠ ${data.status}`) + versionSuffix);
|
|
37
22
|
}
|
|
38
23
|
});
|
|
@@ -1,138 +1,18 @@
|
|
|
1
1
|
import { Command } from 'commander';
|
|
2
|
-
/**
|
|
3
|
-
* #108 — surface reorg. `run`/`list`/`get`/`data`/`history`/`run-info`/
|
|
4
|
-
* `trigger` are promoted to top-level verbs (see index.ts's `createProgram`)
|
|
5
|
-
* alongside `create`/`whoami`/`ping` (#114 — `create` replaced `fetch` in
|
|
6
|
-
* that group). Each is built by an exported
|
|
7
|
-
* `attachXCommand(parent, attachOpts)` factory instead of a fixed
|
|
8
|
-
* `scraps.command(...)` chain, so it can be attached TWICE with a single
|
|
9
|
-
* source-of-truth definition: once to `program` (the new canonical
|
|
10
|
-
* top-level path) and once more, hidden, right back onto `scraps` — so every
|
|
11
|
-
* pre-#108 `trawl scraps <verb>` invocation keeps resolving unchanged
|
|
12
|
-
* (`{ hidden: true }` only affects help visibility, never resolution).
|
|
13
|
-
*
|
|
14
|
-
* Future-drift guard: ALL `.option()`/`.argument()`/`.action()` wiring for a
|
|
15
|
-
* promoted verb MUST live INSIDE its `attachXCommand` factory body, never
|
|
16
|
-
* bolted onto one of its two call sites (index.ts's `createProgram` for the
|
|
17
|
-
* top-level attach, this file's own `attachXCommand(scraps, { hidden: true
|
|
18
|
-
* })` call for the legacy one). That's the only thing keeping `trawl run
|
|
19
|
-
* <id>` and `trawl scraps run <id>` identical — a call-site-only tweak to
|
|
20
|
-
* one attachment would silently diverge the two.
|
|
21
|
-
*/
|
|
22
2
|
type AttachOptions = {
|
|
23
3
|
hidden?: boolean;
|
|
24
4
|
};
|
|
25
5
|
export declare const scraps: Command;
|
|
26
|
-
/** The top-of-history snapshot pollRunProgress needs to identify which run
|
|
27
|
-
* it's watching — see captureBeforeRunState.
|
|
28
|
-
*
|
|
29
|
-
* #97 — `captured` discriminates WHY `id` is undefined: `true` means the GET
|
|
30
|
-
* succeeded and the scrap genuinely has no history yet (an honest "never
|
|
31
|
-
* run" signal pollRunProgress can trust immediately); `false` means the GET
|
|
32
|
-
* itself threw, so `id`/`alreadyInFlight` carry NO information at all — the
|
|
33
|
-
* scrap could easily have prior (possibly terminal) history that this
|
|
34
|
-
* lookup simply never saw. Before this field existed, both cases produced
|
|
35
|
-
* the identical `{id: undefined, alreadyInFlight: false}` shape, so
|
|
36
|
-
* pollRunProgress could not tell them apart (see its own #97 comment).
|
|
37
|
-
*/
|
|
38
6
|
interface BeforeRunState {
|
|
39
7
|
id?: string;
|
|
40
8
|
alreadyInFlight: boolean;
|
|
41
9
|
captured: boolean;
|
|
42
10
|
}
|
|
43
|
-
/** The machine-readable outcome `pollRunProgress` prints as the single final
|
|
44
|
-
* NDJSON line under `--json` (#107 review F1). `status` is the same honest
|
|
45
|
-
* terminal string the human-mode "Run finished: <status>" line already
|
|
46
|
-
* shows (`success`/`error`/`empty`/`regression`/…), or `'timeout'` /
|
|
47
|
-
* `'poll_error'` for the two non-terminal exits. */
|
|
48
11
|
export interface PollOutcome {
|
|
49
12
|
runId?: string;
|
|
50
13
|
status: string;
|
|
51
|
-
/** Present only for `status:'poll_error'` — the last poll failure's message. */
|
|
52
14
|
error?: string;
|
|
53
15
|
}
|
|
54
|
-
/**
|
|
55
|
-
* #91 P1 — replaces "await the run to completion, THEN open the activities
|
|
56
|
-
* SSE stream" (which showed NOTHING: the activities SSE
|
|
57
|
-
* (GET /api/scraps/:id/activities/stream) is backed by an in-process
|
|
58
|
-
* EventEmitter with no backlog — trawl_node
|
|
59
|
-
* modules/activities/services/activities.service.js — so by the time a
|
|
60
|
-
* synchronous run has already finished there is nothing left to emit; and
|
|
61
|
-
* the async `trigger` default enqueues the run onto a durable job queue a
|
|
62
|
-
* SEPARATE cron-consumer pod drains, whose in-process emitter never reaches
|
|
63
|
-
* the API pod holding the SSE connection at all).
|
|
64
|
-
*
|
|
65
|
-
* Instead this polls two REST reads that are BOTH Mongo-backed (not
|
|
66
|
-
* in-process), so they work no matter which pod actually executed the run:
|
|
67
|
-
* - GET /api/scraps/:id/activities?history=<hid>&limit=20 — the SAME
|
|
68
|
-
* activities-list endpoint `doctor`'s fetchRunAndFix already calls
|
|
69
|
-
* (src/commands/doctor.ts) — prints each new activity line once the new
|
|
70
|
-
* run's history id is known.
|
|
71
|
-
* - GET /api/scraps/:id — history[0].status/statusDetail, the SAME
|
|
72
|
-
* terminal-status signal `lastStatus()` above already trusts (status:null
|
|
73
|
-
* === in flight, #88 item 1) to know when the run is done.
|
|
74
|
-
*
|
|
75
|
-
* #93 item 1 — dedup-race fix. "Is this the run we're watching?" used to be
|
|
76
|
-
* a single check: `history[0]._id !== beforeHistoryId`. That's wrong when
|
|
77
|
-
* `before.alreadyInFlight` is true (a `trigger` call deduped onto a worker
|
|
78
|
-
* job that was ALREADY pending/running at capture time): the top row IS the
|
|
79
|
-
* run we're watching, but its `_id` never changes, so the old guard never
|
|
80
|
-
* released and the poll ran the full timeout to a false "Timed out". The run
|
|
81
|
-
* we're watching is now EITHER a brand-new id (fresh trigger, the common
|
|
82
|
-
* case) OR the same id that was already in-flight (status:null) at capture
|
|
83
|
-
* (the dedup case) — a same-id row that was already TERMINAL at capture is
|
|
84
|
-
* neither, and must not be latched onto as "done" (it's just the previous
|
|
85
|
-
* run, still sitting there until a genuinely new run supersedes it).
|
|
86
|
-
*
|
|
87
|
-
* #97 — capture-failed fallback. The dedup-race fix above assumes `before`
|
|
88
|
-
* is trustworthy. When `captureBeforeRunState`'s own GET threw,
|
|
89
|
-
* `before.id` is `undefined` — and that is INDISTINGUISHABLE from "the
|
|
90
|
-
* scrap has genuinely never run" (also `id: undefined`), which is exactly
|
|
91
|
-
* the case `isNewRun` above is designed to match on the very first row that
|
|
92
|
-
* ever appears. So on the very first poll, ANY pre-existing history row —
|
|
93
|
-
* even the STALE PREVIOUS run, already terminal — satisfied
|
|
94
|
-
* `last._id !== undefined` and got reported as "the run we just launched"
|
|
95
|
-
* finishing, when it was really just whatever ran before.
|
|
96
|
-
*
|
|
97
|
-
* Fix: `before.captured === false` defers trusting a baseline at all.
|
|
98
|
-
* Instead of comparing against the (unknown) `beforeId` from the start, the
|
|
99
|
-
* FIRST successful poll read is treated as the deferred capture itself —
|
|
100
|
-
* exactly what `captureBeforeRunState` would have returned had its GET
|
|
101
|
-
* succeeded — and only READS from that point on are compared against it,
|
|
102
|
-
* via the exact same `isNewRun` / `isDedupOntoInFlight` logic above. A
|
|
103
|
-
* pre-existing terminal row observed on that first read becomes `beforeId`
|
|
104
|
-
* (not a match for itself), so it correctly falls into "still the stale
|
|
105
|
-
* previous run" below and the poll keeps waiting; a row still in flight
|
|
106
|
-
* becomes the `alreadyInFlight` baseline, exactly like a successful capture
|
|
107
|
-
* would have recorded. Either way this costs at most one extra poll
|
|
108
|
-
* interval, bounded by the same deadline as everything else. The one
|
|
109
|
-
* remaining edge case — capture failed AND the scrap never ran before AND
|
|
110
|
-
* the triggered run already finished by the very first poll — is genuinely
|
|
111
|
-
* undecidable from "stale pre-existing row" with no more information than
|
|
112
|
-
* this function has, so it resolves to the same honest timeout rather than
|
|
113
|
-
* risk reporting a possibly-wrong outcome (never a lie, at worst a timeout
|
|
114
|
-
* telling the caller to check `doctor`).
|
|
115
|
-
*
|
|
116
|
-
* #107 review F1 — before this fix, `run|trigger --json --watch` was
|
|
117
|
-
* outcome-blind: `quiet` suppressed ALL output (including "Run finished:
|
|
118
|
-
* failure" and the timeout notice), a transient poll error was caught and
|
|
119
|
-
* silently retried FOREVER within the deadline, and the process always
|
|
120
|
-
* exited 0 after the poll loop regardless of what the watched run actually
|
|
121
|
-
* did — dead air, then a clean exit code, even for a failed or timed-out
|
|
122
|
-
* run. An agent scripting this CLI had no way to tell success from failure
|
|
123
|
-
* from "we gave up". Fixed by:
|
|
124
|
-
* - emitting exactly ONE final NDJSON line on stdout under `--json` once
|
|
125
|
-
* the watch reaches ANY of its three exits (terminal status, timeout, or
|
|
126
|
-
* a persistent poll error) — `{runId,status}` (+`error` for a poll
|
|
127
|
-
* error) — while every intermediate progress line stays suppressed
|
|
128
|
-
* (unchanged from before);
|
|
129
|
-
* - setting `process.exitCode` non-zero on a genuine run failure, a
|
|
130
|
-
* timeout, or a persistent poll error, and `0` on a real success — in
|
|
131
|
-
* BOTH `--json` and human `--watch` modes (human mode used to exit 0
|
|
132
|
-
* unconditionally, the same bug, just silent instead of dishonest);
|
|
133
|
-
* - giving up after `MAX_CONSECUTIVE_POLL_ERRORS` consecutive failed reads
|
|
134
|
-
* instead of retrying the same dead endpoint for the full 300s.
|
|
135
|
-
*/
|
|
136
16
|
export declare function pollRunProgress(id: string, before: BeforeRunState | undefined, opts?: {
|
|
137
17
|
intervalMs?: number;
|
|
138
18
|
timeoutMs?: number;
|