@trawlme/cli 3.12.0 → 3.12.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (73) hide show
  1. package/README.md +1 -1
  2. package/dist/commands/create.d.ts +0 -28
  3. package/dist/commands/create.js +0 -89
  4. package/dist/commands/doctor.d.ts +0 -79
  5. package/dist/commands/doctor.js +1 -187
  6. package/dist/commands/login.js +0 -67
  7. package/dist/commands/ping.d.ts +0 -15
  8. package/dist/commands/ping.js +0 -15
  9. package/dist/commands/scraps.d.ts +0 -120
  10. package/dist/commands/scraps.js +10 -724
  11. package/dist/commands/skills.js +0 -22
  12. package/dist/commands/spec.d.ts +0 -85
  13. package/dist/commands/spec.js +0 -67
  14. package/dist/commands/telemetry.js +0 -4
  15. package/dist/commands/token.js +0 -28
  16. package/dist/commands/upgrade.js +0 -22
  17. package/dist/commands/whoami.d.ts +0 -12
  18. package/dist/commands/whoami.js +0 -6
  19. package/dist/index.d.ts +0 -188
  20. package/dist/index.js +0 -349
  21. package/dist/lib/api.d.ts +0 -78
  22. package/dist/lib/api.js +1 -320
  23. package/dist/lib/cdp-pipe.d.ts +0 -72
  24. package/dist/lib/cdp-pipe.js +1 -81
  25. package/dist/lib/chrome-discovery.d.ts +0 -11
  26. package/dist/lib/chrome-discovery.js +0 -19
  27. package/dist/lib/chrome-launch.d.ts +0 -40
  28. package/dist/lib/chrome-launch.js +0 -69
  29. package/dist/lib/config.d.ts +0 -53
  30. package/dist/lib/config.js +0 -55
  31. package/dist/lib/confirm.d.ts +0 -55
  32. package/dist/lib/confirm.js +0 -47
  33. package/dist/lib/docs.d.ts +0 -123
  34. package/dist/lib/docs.js +0 -169
  35. package/dist/lib/errors.d.ts +0 -134
  36. package/dist/lib/errors.js +0 -151
  37. package/dist/lib/format.d.ts +0 -6
  38. package/dist/lib/format.js +0 -6
  39. package/dist/lib/json.d.ts +0 -35
  40. package/dist/lib/json.js +0 -48
  41. package/dist/lib/jwt.d.ts +0 -7
  42. package/dist/lib/jwt.js +0 -7
  43. package/dist/lib/pinch.d.ts +0 -53
  44. package/dist/lib/pinch.js +6 -112
  45. package/dist/lib/pinchAnimation.d.ts +0 -16
  46. package/dist/lib/pinchAnimation.js +8 -29
  47. package/dist/lib/posthog.d.ts +0 -9
  48. package/dist/lib/posthog.js +0 -23
  49. package/dist/lib/prompt.js +1 -20
  50. package/dist/lib/secure-transport.d.ts +0 -7
  51. package/dist/lib/secure-transport.js +0 -24
  52. package/dist/lib/session-capture-guard.d.ts +0 -15
  53. package/dist/lib/session-capture-guard.js +0 -5
  54. package/dist/lib/session-capture.d.ts +0 -125
  55. package/dist/lib/session-capture.js +0 -281
  56. package/dist/lib/skills.d.ts +0 -175
  57. package/dist/lib/skills.js +1 -216
  58. package/dist/lib/skillsNudge.d.ts +0 -17
  59. package/dist/lib/skillsNudge.js +0 -83
  60. package/dist/lib/spinner.d.ts +0 -39
  61. package/dist/lib/spinner.js +0 -40
  62. package/dist/lib/storage-state.d.ts +0 -112
  63. package/dist/lib/storage-state.js +0 -131
  64. package/dist/lib/tips.d.ts +0 -38
  65. package/dist/lib/tips.js +0 -77
  66. package/dist/lib/updateCheckWorker.js +0 -14
  67. package/dist/lib/updateNotifier.d.ts +0 -17
  68. package/dist/lib/updateNotifier.js +0 -53
  69. package/dist/lib/validate.d.ts +0 -8
  70. package/dist/lib/validate.js +0 -8
  71. package/dist/lib/version.d.ts +0 -12
  72. package/dist/lib/version.js +1 -13
  73. package/package.json +2 -2
package/README.md CHANGED
@@ -36,7 +36,7 @@ Four methods:
36
36
  3. **Token flag** — `trawl login --token <jwt>` (CI/CD, direct JWT)
37
37
  4. **API key** — `TRAWL_API_KEY=trawl_xxx trawl list` (scoped, revocable one-by-one — the recommended credential for an agent driving this CLI; see [docs/agent-quickstart.md](docs/agent-quickstart.md))
38
38
 
39
- A `TRAWL_API_KEY` is a `trawl_*`-prefixed credential (create/revoke one in the Trawl dashboard) sent as `Authorization: Bearer` instead of the session `Cookie: TOKEN=` a JWT uses — `trawl` picks the right one automatically based on the credential's own shape, never a flag. `TRAWL_API_KEY` wins over `TRAWL_TOKEN` when both happen to be set. It is scoped server-side — in practice to a set of scraps, since the dashboard's key-create form sets only the scrap allow-list; narrowing which *actions* a key may perform needs an explicit `scopes` array at creation over the API, and a key without one has every action granted. It works on most of the Core tier — `create`/`list`/`get`/`data`/`history`/`run-info`/`run`/`trigger`/`ping`, including the `--watch` **polling flag** on `run`/`trigger` — plus, outside Core, `scraps account status`/`scraps doctor`/`scraps autofix` (all three only ever read routes trawl_node opened to keys). It does **not** work on `whoami`, `scraps update`/`delete`, `scraps account set`/`delete`/`clear-session`/`session set`, `scraps banner`, `scraps snapshot` (the scrap lookup it starts from is dual-auth, but the `html-snapshot` route it downloads from isn't), or the standalone SSE **command** `scraps watch` (do not conflate the two: `--watch` is a flag on `run`/`trigger` and works under a key; `scraps watch` is a separate command and is JWT-only) — those stay JWT-only and fail with a `kind:"auth"` envelope (exit `3`) pointing at `trawl login` under a key. This list mirrors trawl_node's route wiring as of this writing, not a frozen guarantee — for anything not named here, trust the real `--json` envelope over this paragraph. Never send both a key and a JWT on the same request — the CLI only ever attaches one. `list --unhealthy` (and the plain health badge on `list`/`get`) rides the same `GET /api/scraps`/`GET /api/scraps/:id` routes as the rest of the group — no separate auth path — so it works under a key exactly like plain `list`/`get`, and a scrap-scoped key still only ever sees its own allow-listed scraps through the filter.
39
+ A `TRAWL_API_KEY` is a `trawl_*`-prefixed credential (create/revoke one in the Trawl dashboard) sent as `Authorization: Bearer` instead of the session `Cookie: TOKEN=` a JWT uses — `trawl` picks the right one automatically based on the credential's own shape, never a flag. `TRAWL_API_KEY` wins over `TRAWL_TOKEN` when both happen to be set. It is scoped server-side — in practice to a set of scraps, since the dashboard's key-create form sets only the scrap allow-list; narrowing which *actions* a key may perform needs an explicit `scopes` array at creation over the API, and a key without one has every action granted. It works on most of the Core tier — `create`/`list`/`get`/`data`/`history`/`run-info`/`run`/`trigger`/`ping`, including the `--watch` **polling flag** on `run`/`trigger` — plus, outside Core, `scraps account status`/`scraps doctor`/`scraps autofix` (all three only ever read routes trawl_node opened to keys). It does **not** work on `whoami`, `scraps update`/`delete`, `scraps account set`/`delete`/`clear-session`/`session set`/`session capture`, `scraps banner`, `scraps snapshot` (the scrap lookup it starts from is dual-auth, but the `html-snapshot` route it downloads from isn't), or the standalone SSE **command** `scraps watch` (do not conflate the two: `--watch` is a flag on `run`/`trigger` and works under a key; `scraps watch` is a separate command and is JWT-only) — those stay JWT-only and fail with a `kind:"auth"` envelope (exit `3`) pointing at `trawl login` under a key. This list mirrors trawl_node's route wiring as of this writing, not a frozen guarantee — for anything not named here, trust the real `--json` envelope over this paragraph. Never send both a key and a JWT on the same request — the CLI only ever attaches one. `list --unhealthy` (and the plain health badge on `list`/`get`) rides the same `GET /api/scraps`/`GET /api/scraps/:id` routes as the rest of the group — no separate auth path — so it works under a key exactly like plain `list`/`get`, and a scrap-scoped key still only ever sees its own allow-listed scraps through the filter.
40
40
 
41
41
  Custom API URL: `trawl login --url https://self-hosted.example.com`
42
42
 
@@ -1,37 +1,9 @@
1
1
  import { Command } from 'commander';
2
- /**
3
- * `POST /api/ai/wizard` response contract (#114, trawl_node —
4
- * ai.wizard.service.js#runWizard, shipped S1, contract LOCKED). The wizard
5
- * chains, entirely server-side: AI code generation -> scrap creation ->
6
- * the FIRST run (ScrapsService.load) -> auto-fix on failure (autoFix
7
- * defaults true unless `--no-autofix` maps to `autoFix:false`).
8
- *
9
- * - `success` is an HONEST outcome of that first run, not "did the HTTP
10
- * call succeed" — a failed first run is still a 200 response (the scrap
11
- * itself was created either way; auto-fix, when enabled, retries in the
12
- * background). Callers must branch on `success`, never assume 2xx means
13
- * "the scrap works".
14
- * - `scrap` is the full created Scrap object (present whenever creation got
15
- * far enough to persist it — a hard failure before that point surfaces as
16
- * a real HTTP error instead, handled by the shared error path).
17
- * - `historyId` is best-effort (a failed server-side lookup leaves it
18
- * `null`, never breaks the response) — it points at the first run just
19
- * executed.
20
- */
21
2
  export interface WizardResponse {
22
3
  success: boolean;
23
4
  scrap?: {
24
5
  _id: string;
25
6
  title: string;
26
- /**
27
- * Cron expression the wizard schedules this scrap on. As of #114/S1
28
- * (trawl_node ai.wizard.service.js#runWizard's `scrapBody`) this is
29
- * hardcoded server-side to a DAILY run — `'0 7 * * *'` / `cronTimezone:
30
- * 'UTC'` — unconditionally, regardless of `--prompt`/`--no-autofix`.
31
- * Present on the scrap object returned here (the wizard controller
32
- * passes the created scrap straight through, no stripping) — change or
33
- * disable it with `trawl scraps update <id> --cron <expr>` / `--no-cron`.
34
- */
35
7
  cron?: string | null;
36
8
  cronTimezone?: string;
37
9
  [key: string]: unknown;
@@ -8,12 +8,6 @@ import { requireUrl, requireString } from '../lib/validate.js';
8
8
  import { UsageError } from '../lib/errors.js';
9
9
  import { renderPinch, pinchEnabled } from '../lib/pinch.js';
10
10
  import { startPinchAnimation } from '../lib/pinchAnimation.js';
11
- /** Best-effort, honest first-run summary — never claims a background retry
12
- * happened when auto-fix was disabled for this call, and never claims a
13
- * scrap was persisted when the response carries none (#114-F3 — a hard
14
- * failure before persistence still comes back as `success:false` with no
15
- * `scrap` at all; claiming "auto-fix retrying in the background" then would
16
- * be fabricated — there is nothing to retry). */
17
11
  function firstRunLabel(data, autoFixEnabled) {
18
12
  if (data.success)
19
13
  return 'succeeded';
@@ -23,10 +17,6 @@ function firstRunLabel(data, autoFixEnabled) {
23
17
  return 'failed (auto-fix retrying in the background)';
24
18
  return 'failed';
25
19
  }
26
- /** Best-effort human description of a daily cron (`M H * * *`) — the only
27
- * shape the wizard's server-side default currently produces. Falls back to
28
- * printing the raw expression for anything else rather than guessing at a
29
- * schedule the CLI can't actually parse. */
30
20
  function describeCron(cron) {
31
21
  const match = /^(\d{1,2})\s+(\d{1,2})\s+\*\s+\*\s+\*$/.exec(cron.trim());
32
22
  if (!match)
@@ -34,16 +24,6 @@ function describeCron(cron) {
34
24
  const [, min, hour] = match;
35
25
  return `daily ${hour.padStart(2, '0')}:${min.padStart(2, '0')}`;
36
26
  }
37
- /**
38
- * #114-F2 — a wizard-created scrap runs on a DAILY cron by default
39
- * server-side; surface that up front rather than leaving it to be
40
- * discovered later as an unexpected recurring quota charge. Reads the real
41
- * `cron`/`cronTimezone` field off the response scrap when present; falls
42
- * back to the known wizard default wording ONLY when the field is missing
43
- * from the response (an older server, or a future rename) — never invents a
44
- * schedule value that might not match what the server actually applied.
45
- * Returns null when there's no scrap to schedule at all.
46
- */
47
27
  function scheduleLabel(scrap) {
48
28
  if (!scrap)
49
29
  return null;
@@ -53,24 +33,11 @@ function scheduleLabel(scrap) {
53
33
  }
54
34
  return 'scheduled daily by default';
55
35
  }
56
- /**
57
- * #121 — onboarding principle: a successful `create` should SHOW the value,
58
- * not just an id. Best-effort fetch of the first run's persisted data (the
59
- * same read-only, no-quota path `trawl data <id>` uses:
60
- * `GET /api/historys/:historyId` → a JSON string `{ data: [...] }`) and print
61
- * a small proof-of-value sample (count + first-item keys + one truncated
62
- * value line — never a full dump; that's what `trawl data --json` is for).
63
- * ANY failure (network, parse, no data) is swallowed silently — the sample is
64
- * a bonus, it must never turn a successful create into a failure or noise.
65
- */
66
36
  async function printDataSample(historyId) {
67
37
  try {
68
38
  const detail = await api.get(`/api/historys/${historyId}`);
69
39
  if (typeof detail?.data !== 'string' || !detail.data)
70
40
  return;
71
- // #159 — same JSON-string-within-JSON shape as scrap.history[0].data;
72
- // tolerate the documented raw-control-char server quirk here too instead
73
- // of silently losing the sample to the outer try/catch below.
74
41
  const items = parseServerJson(detail.data)?.data;
75
42
  if (!Array.isArray(items) || items.length === 0)
76
43
  return;
@@ -88,7 +55,6 @@ async function printDataSample(historyId) {
88
55
  }
89
56
  }
90
57
  catch {
91
- // best-effort — a missing sample never fails or noises up a good create
92
58
  }
93
59
  }
94
60
  export const create = new Command('create')
@@ -99,15 +65,6 @@ export const create = new Command('create')
99
65
  .option('--no-autofix', 'Disable AI auto-fix on first-run failure (default: on)')
100
66
  .option('--json', 'Output the raw API payload')
101
67
  .action(async (rawUrl, opts) => {
102
- // Fast, local usage-errors (exit 2) — never a round-trip to the server
103
- // for something we can already tell is bad. Same fail-fast pattern the
104
- // former `fetch` command used for its URL argument.
105
- //
106
- // #116 — the url is accepted BOTH as the positional argument (agent
107
- // one-liner) and as `--url` (muscle memory from `scraps create` and the
108
- // API body {url, goal}). Exactly one is required; both are fine only
109
- // when identical — two DIFFERENT urls is ambiguous, refuse loudly
110
- // rather than silently picking one.
111
68
  if (rawUrl === undefined && opts.url === undefined) {
112
69
  throw new UsageError("missing required argument 'url' (positional, or --url <url>)");
113
70
  }
@@ -121,58 +78,28 @@ export const create = new Command('create')
121
78
  goal,
122
79
  ...(opts.autofix === false && { autoFix: false }),
123
80
  };
124
- // #91/#106-F1 long-run pattern — the wizard runs AI generation + scrap
125
- // creation + a real FIRST run (+ autofix retries) entirely server-side,
126
- // legitimately 30-250s+. The 30s DEFAULT_TIMEOUT_MS would abort it
127
- // mid-flight and fabricate a NetworkError timeout for a request that was
128
- // always going to succeed.
129
81
  const call = () => api.post('/api/ai/wizard', body, { timeoutMs: LONG_RUN_TIMEOUT_MS });
130
- // #106-F2 pattern carried over from `fetch` — under --json stdout must
131
- // be provably pure: no spinner channel at all. Only the human path gets
132
- // the ora progress indicator; --json calls the API directly.
133
82
  let data;
134
83
  if (opts.json) {
135
84
  data = await call();
136
85
  }
137
86
  else {
138
- // #122/#131 — while the server-side wizard runs (legitimately 30-250s+,
139
- // see LONG_RUN_TIMEOUT_MS above) Pinch ANIMATES in place (claw-wiggle)
140
- // on a color-capable TTY; otherwise the ora spinner (which itself
141
- // no-ops under a non-TTY, so piped human output stays clean). Both
142
- // write to stderr only — stdout purity under `--json` is guaranteed by
143
- // the `opts.json` branch above, never by this code.
144
87
  const anim = pinchEnabled() ? startPinchAnimation(`Creating a scrap from ${url}…`) : null;
145
88
  try {
146
89
  data = anim
147
90
  ? await call()
148
91
  : await spin(call, {
149
- // No verdict symbol (#106-F3) — ora's success only means "the
150
- // HTTP call didn't throw", not "the first run succeeded".
151
92
  text: `Creating a scrap from ${url}…`,
152
93
  successText: 'Request complete',
153
94
  });
154
95
  }
155
96
  catch (err) {
156
97
  anim?.stop();
157
- // #114-F1 — a client-side timeout (NetworkError, "timed out after
158
- // …ms" per api.ts's safeFetch) does NOT mean the wizard failed
159
- // server-side: the scrap creation + first run keep going on the
160
- // server after the CLI gives up waiting, so the scrap may already
161
- // exist (or land moments later). Warn BEFORE rethrowing so a retry
162
- // isn't the first instinct — a blind retry creates a duplicate scrap
163
- // and burns AI-generation quota a second time for the same goal.
164
- // Only a NetworkError whose message identifies it as the timeout
165
- // branch qualifies — a DNS/connection-refused NetworkError never
166
- // reached the server at all, so there's nothing to warn about here.
167
98
  if (err instanceof NetworkError && /timed out/i.test(err.message)) {
168
99
  console.error(chalk.yellow('⚠ The request timed out client-side, but the scrap may STILL have been created server-side — run `trawl list` before retrying (a retry creates a DUPLICATE scrap + burns quota).'));
169
100
  }
170
- // Rethrow unchanged so index.ts's classifyError/exit-code taxonomy
171
- // stays intact (this stays a NetworkError -> exit 5, same as before).
172
101
  throw err;
173
102
  }
174
- // Success — stop + erase the animation so the result prints on a clean
175
- // line (the celebrating/confused frame below is the one-shot outcome).
176
103
  anim?.stop();
177
104
  }
178
105
  if (opts.json) {
@@ -190,19 +117,11 @@ export const create = new Command('create')
190
117
  console.log(chalk.dim(` First run: `) + firstRunLabel(data, autoFixEnabled));
191
118
  if (schedule)
192
119
  console.log(chalk.dim(` Schedule: `) + schedule);
193
- // #121 — on a successful first run, prove the value: show a small data
194
- // sample (best-effort, silent on failure) before the next-step hint.
195
120
  if (data.success && data.historyId)
196
121
  await printDataSample(data.historyId);
197
- // #121 — the natural next command is the DATA, not the metadata: point
198
- // at `trawl data` primarily, keep `trawl get` as the secondary detail view.
199
122
  if (scrapId) {
200
123
  console.log(chalk.dim(` Next step: `) + `trawl data ${scrapId}` + chalk.dim(` (details: trawl get ${scrapId})`));
201
124
  }
202
- // #114-F3 — only claim a background retry is happening when a scrap
203
- // actually exists to retry (never fabricate progress that isn't real);
204
- // `firstRunLabel` above already covers the !scrap / autofix-disabled
205
- // wording, this adds the actionable poll target on top.
206
125
  if (!data.success && data.scrap && autoFixEnabled) {
207
126
  const pollTarget = data.historyId
208
127
  ? `\`trawl data ${scrapId}\` or \`trawl run-info ${data.historyId}\``
@@ -210,18 +129,10 @@ export const create = new Command('create')
210
129
  console.log(chalk.yellow(` Note: `) +
211
130
  `Auto-fix is retrying in the background — do NOT re-run create; poll ${pollTarget}.`);
212
131
  }
213
- // #122 — Pinch reacts to the HONEST first-run outcome (same signal the
214
- // ✓/✗ line above already renders): celebrates a real success, looks
215
- // confused on a genuine failure. Belt-and-suspenders `!opts.json`
216
- // alongside pinchEnabled() — see the 'thinking' print above for why.
217
132
  if (!opts.json && pinchEnabled()) {
218
133
  console.log(data.success ? renderPinch('celebrating') : renderPinch('confused'));
219
134
  }
220
135
  }
221
- // Honest exit code alongside the honest payload — a --json caller gets
222
- // the raw body regardless (never wrapped/altered), but a script checking
223
- // the exit code alone must be able to tell "first run failed" from "ran
224
- // fine" without parsing. Same contract the former `fetch` command used.
225
136
  if (!data.success)
226
137
  process.exitCode = 1;
227
138
  });
@@ -1,9 +1,3 @@
1
- /**
2
- * Run diagnostics shape — mirrors the owner-safe REST shape returned by
3
- * GET /api/historys/:id (Phase 1 projection, 2026-05-27).
4
- * Cost and proxy-nature fields are intentionally absent by design.
5
- * Abstract proxyTier (Tier 0–4) is kept.
6
- */
7
1
  export interface Run {
8
2
  _id: string;
9
3
  status: boolean | null;
@@ -28,11 +22,6 @@ export interface Run {
28
22
  block?: {
29
23
  kind?: string | null;
30
24
  } | null;
31
- /** @deprecated trawl_node#1950 renamed this to `block.kind`. Kept as a read
32
- * fallback (see detectWallVendor) for the window where this CLI is
33
- * published ahead of the trawl_node prod tag — a prod backend served from
34
- * the pre-#1950 tag still returns this flat field, not `block.kind`. Drop
35
- * once prod is confirmed on a tag containing #1950. */
36
25
  blockType?: string | null;
37
26
  proxyTier?: string | null;
38
27
  baselineLength?: number | null;
@@ -40,64 +29,14 @@ export interface Run {
40
29
  createdAt?: string;
41
30
  time?: number | null;
42
31
  triggeredBy?: string | null;
43
- /** trawl_node#1975 — freeform failure classification (`'auth'` = login-wall
44
- * empty run, cookies are the fix). Not a TS union — trawl_node's own set
45
- * is additive/open, so this CLI must stay read-safe against a future
46
- * value it doesn't know about yet, same posture as `block.kind` above. */
47
32
  failureKind?: string | null;
48
33
  }
49
- /**
50
- * Resolve a known anti-bot vendor name from a worker `block.kind`
51
- * string. Returns null when the run succeeded, when there is no `block.kind`
52
- * signal at all, or when `block.kind` names something other than a known
53
- * vendor (e.g. `proxy-domain-gate`, `rate_limited_per_host`) — those stay on
54
- * the genuine-error path since we can't honestly attribute them to a specific
55
- * "no reliable bypass" wall.
56
- *
57
- * `blocked` is deliberately NOT consulted here (#177): it is a mid-run,
58
- * attempt-level signal the worker only stamps on one envelope shape, so it
59
- * reads `false` on the great majority of genuinely walled runs (the early-block
60
- * throw path sets `block.kind` but never `blocked` — 63 of 64 walled runs
61
- * measured on prod). `status === true` wins instead: a run that ultimately
62
- * returned data was not walled, even if an earlier tier's `block.kind` stamp
63
- * survived on the row.
64
- *
65
- * trawl_node#1950 renamed the flat `blockType` field to nested `block.kind`.
66
- * Read `block?.kind` first, falling back to the deprecated flat `blockType` —
67
- * this CLI can be published (and talk to prod) before the trawl_node prod tag
68
- * containing #1950 is cut (see the `Run.blockType` doc comment), so a prod
69
- * response can still be the pre-#1950 flat shape for a while.
70
- */
71
34
  export declare function detectWallVendor(run: Pick<Run, 'status' | 'statusDetail' | 'block' | 'blockType'>): string | null;
72
- /**
73
- * #184 — true for a run currently carrying a live (non-stale) login-wall
74
- * verdict. Extracted out of `formatDoctor`'s own local `authWall` const so
75
- * `commands/scraps.ts`'s `doctor`/`run-info` actions can reuse the EXACT
76
- * same guard to decide whether to fire the skills-install safety-net nudge
77
- * (lib/skillsNudge.ts `maybeSuggestSkillsForAuthWall`) — one rule, not two
78
- * copies that could quietly drift apart.
79
- *
80
- * Same staleness guard as `detectWallVendor` above and for the same reason:
81
- * `failureKind` is a terminal classification trawl_node stamps once, but
82
- * `patchForRegression` can flip `status`/`statusDetail` to success/
83
- * regression LATER without ever clearing it — so a run that ultimately
84
- * succeeded or degraded must never still read as an active auth wall.
85
- *
86
- * Typed structurally loose (not `Pick<Run, ...>`) on purpose: `Run.status`/
87
- * `statusDetail` are required fields, but `scraps.ts`'s own `HistoryRun`
88
- * (the run-info command's shape) declares the same three fields OPTIONAL —
89
- * a `Pick<Run, ...>` parameter type would reject that caller at compile
90
- * time even though every field it actually reads is present at runtime.
91
- */
92
35
  export declare function isAuthWall(run: {
93
36
  failureKind?: string | null;
94
37
  status?: boolean | null;
95
38
  statusDetail?: string | null;
96
39
  }): boolean;
97
- /**
98
- * Autofix activity metadata — from the persisted ai_fix_end activity.
99
- * aiUsage (cost) is stripped server-side; all diagnostics are kept.
100
- */
101
40
  export interface FixActivity {
102
41
  outcome: 'applied' | 'failed' | 'skipped' | 'breaker_tripped' | 'timeout' | null;
103
42
  classification?: string | null;
@@ -127,26 +66,8 @@ interface ScrapHead {
127
66
  }
128
67
  export declare function pickRun(run: Run): Partial<Run>;
129
68
  export declare function pickFix(fix: FixActivity | null): Partial<FixActivity> | null;
130
- /**
131
- * Pure formatter — returns a human-readable diagnosis string for a run.
132
- * Used by `doctor`, `doctor --autofix`, and `data --errors`.
133
- *
134
- * @param scrapTitle - Display name of the scrap
135
- * @param run - Owner-safe history run object
136
- * @param fix - Optional autofix activity (null = no fix attempted)
137
- * @param scrapId - Scrap document id (used in hint lines — autofix/snapshot take scrap id, not run id)
138
- * @returns Multi-line string ready for console.log
139
- */
140
69
  export declare function formatDoctor(scrapTitle: string, run: Run, fix?: FixActivity | null, scrapId?: string): string;
141
- /**
142
- * Pure formatter — returns a human-readable autofix detail block.
143
- * Shows diff, dry-run results, knowledge consulted.
144
- */
145
70
  export declare function formatAutofix(fix: FixActivity): string;
146
- /**
147
- * Fetch the latest run + its ai_fix_end activity for a scrap.
148
- * Returns null if the scrap has no runs yet.
149
- */
150
71
  export declare function fetchRunAndFix(scrapId: string): Promise<{
151
72
  scrap: ScrapHead;
152
73
  run: Run;
@@ -1,27 +1,6 @@
1
1
  import { api } from '../lib/api.js';
2
2
  import chalk from 'chalk';
3
3
  import { formatDate } from '../lib/format.js';
4
- /**
5
- * Known anti-bot vendors that the worker's `block.kind` field may name
6
- * (ex-flat `blockType`, trawl_node#1950). Ordered by first-match;
7
- * `block.kind` is a freeform worker string, not an enum (see the `Run.block`
8
- * doc comment), so this is a best-effort substring match against the real
9
- * field — never an invented/mocked value.
10
- *
11
- * `datadome`/`perimeterx`/`akamai`/`cloudflare` are literal substrings the
12
- * current worker emits (detect.js). `kasada` is forward-compatible: the issue
13
- * (#62 / wall-registry WS-7) names it as a genuine wall, but the current
14
- * worker vocabulary does not yet emit it — this branch lights up
15
- * automatically if/when a future worker classifier does, without a CLI
16
- * change.
17
- *
18
- * trawl_cli#182 — an `auth` entry used to live here, rendering a login wall
19
- * as "walled: auth — no reliable bypass". Removed: unlike a real anti-bot
20
- * vendor, a login wall's own reliable bypass is the user's session cookies,
21
- * so that verdict was wrong, not just early. `failureKind==='auth'`
22
- * (trawl_node#1975) is the honest signal for a login wall — see the
23
- * dedicated branch in `formatDoctor` below.
24
- */
25
4
  const WALL_VENDOR_PATTERNS = [
26
5
  [/datadome/i, 'DataDome'],
27
6
  [/kasada/i, 'Kasada'],
@@ -29,66 +8,18 @@ const WALL_VENDOR_PATTERNS = [
29
8
  [/akamai/i, 'Akamai'],
30
9
  [/cloudflare/i, 'Cloudflare'],
31
10
  ];
32
- /**
33
- * Resolve a known anti-bot vendor name from a worker `block.kind`
34
- * string. Returns null when the run succeeded, when there is no `block.kind`
35
- * signal at all, or when `block.kind` names something other than a known
36
- * vendor (e.g. `proxy-domain-gate`, `rate_limited_per_host`) — those stay on
37
- * the genuine-error path since we can't honestly attribute them to a specific
38
- * "no reliable bypass" wall.
39
- *
40
- * `blocked` is deliberately NOT consulted here (#177): it is a mid-run,
41
- * attempt-level signal the worker only stamps on one envelope shape, so it
42
- * reads `false` on the great majority of genuinely walled runs (the early-block
43
- * throw path sets `block.kind` but never `blocked` — 63 of 64 walled runs
44
- * measured on prod). `status === true` wins instead: a run that ultimately
45
- * returned data was not walled, even if an earlier tier's `block.kind` stamp
46
- * survived on the row.
47
- *
48
- * trawl_node#1950 renamed the flat `blockType` field to nested `block.kind`.
49
- * Read `block?.kind` first, falling back to the deprecated flat `blockType` —
50
- * this CLI can be published (and talk to prod) before the trawl_node prod tag
51
- * containing #1950 is cut (see the `Run.blockType` doc comment), so a prod
52
- * response can still be the pre-#1950 flat shape for a while.
53
- */
54
11
  export function detectWallVendor(run) {
55
- // A run that returned data was not walled. `status === true` is not the whole
56
- // predicate: trawl_node flips a degraded-but-non-empty run to
57
- // `status:false, statusDetail:'regression'` (scraps.service.js, #1112) via a
58
- // patch that never clears `block.kind` — so keying on `status` alone would
59
- // print "no reliable bypass" next to "Regression: length N vs baseline M",
60
- // which is the same contradiction this guard exists to remove.
61
12
  if (run.status === true || run.statusDetail === 'regression')
62
13
  return null;
63
14
  const kind = run.block?.kind ?? run.blockType ?? null;
64
15
  if (!kind)
65
- return null; // no signal at all
16
+ return null;
66
17
  for (const [pattern, label] of WALL_VENDOR_PATTERNS) {
67
18
  if (pattern.test(kind))
68
19
  return label;
69
20
  }
70
21
  return null;
71
22
  }
72
- /**
73
- * #184 — true for a run currently carrying a live (non-stale) login-wall
74
- * verdict. Extracted out of `formatDoctor`'s own local `authWall` const so
75
- * `commands/scraps.ts`'s `doctor`/`run-info` actions can reuse the EXACT
76
- * same guard to decide whether to fire the skills-install safety-net nudge
77
- * (lib/skillsNudge.ts `maybeSuggestSkillsForAuthWall`) — one rule, not two
78
- * copies that could quietly drift apart.
79
- *
80
- * Same staleness guard as `detectWallVendor` above and for the same reason:
81
- * `failureKind` is a terminal classification trawl_node stamps once, but
82
- * `patchForRegression` can flip `status`/`statusDetail` to success/
83
- * regression LATER without ever clearing it — so a run that ultimately
84
- * succeeded or degraded must never still read as an active auth wall.
85
- *
86
- * Typed structurally loose (not `Pick<Run, ...>`) on purpose: `Run.status`/
87
- * `statusDetail` are required fields, but `scraps.ts`'s own `HistoryRun`
88
- * (the run-info command's shape) declares the same three fields OPTIONAL —
89
- * a `Pick<Run, ...>` parameter type would reject that caller at compile
90
- * time even though every field it actually reads is present at runtime.
91
- */
92
23
  export function isAuthWall(run) {
93
24
  return run.failureKind === 'auth' && run.status !== true && run.statusDetail !== 'regression';
94
25
  }
@@ -120,28 +51,8 @@ export function pickFix(fix) {
120
51
  .filter((k) => k in fix)
121
52
  .map((k) => [k, fix[k]]));
122
53
  }
123
- /**
124
- * Pure formatter — returns a human-readable diagnosis string for a run.
125
- * Used by `doctor`, `doctor --autofix`, and `data --errors`.
126
- *
127
- * @param scrapTitle - Display name of the scrap
128
- * @param run - Owner-safe history run object
129
- * @param fix - Optional autofix activity (null = no fix attempted)
130
- * @param scrapId - Scrap document id (used in hint lines — autofix/snapshot take scrap id, not run id)
131
- * @returns Multi-line string ready for console.log
132
- */
133
54
  export function formatDoctor(scrapTitle, run, fix = null, scrapId) {
134
55
  const lines = [];
135
- // Header + status badge. status:null means a run is IN FLIGHT (node
136
- // persists {status:null, statusDetail:null, inFlight:true} the moment a
137
- // run starts) — that must never render as "failed". (#88 item 1)
138
- //
139
- // #91 LOW — a regression row (status:false, statusDetail:'regression') was
140
- // falling into the same red "failed" bucket as a genuine failure, then
141
- // showing a contradictory "Regression: X vs Y" detail line right below it.
142
- // A regression run's write actually SUCCEEDED (the item count just dropped
143
- // vs baseline, detected by an async patch afterward — see #88 item 2 in
144
- // scraps.ts) — it needs its own distinct badge, not "failed".
145
56
  const ok = run.status === true;
146
57
  const badge = ok
147
58
  ? chalk.green('● success')
@@ -154,69 +65,8 @@ export function formatDoctor(scrapTitle, run, fix = null, scrapId) {
154
65
  : chalk.red('● failed');
155
66
  lines.push(`${chalk.bold(scrapTitle)} ${badge}${run.statusDetail ? ` (${run.statusDetail})` : ''}`);
156
67
  lines.push(chalk.dim(` Run ID: ${run._id}`));
157
- // Error message — an honest accept-wall string for known-walled scraps
158
- // (DataDome/Kasada/PerimeterX/Akamai terminal verdict from the worker),
159
- // otherwise the real error (genuine transient failure).
160
- //
161
- // trawl_cli#182 — the wall-vendor verdict (this CLI's own
162
- // WALL_VENDOR_PATTERNS matched against block.kind/blockType) and the
163
- // login-wall hint (node's failureKind==='auth', trawl_node#1975) are two
164
- // INDEPENDENT server-derived signals with no shared contract: node's block
165
- // classifier (modules/historys/helpers/failureKind.js BLOCK_TYPE_PATTERN)
166
- // and this CLI's vendor table live in different repos and can legitimately
167
- // disagree. Concretely, `kasada` matches this CLI's table but is absent
168
- // from node's block pattern, and the deprecated flat `blockType` fallback
169
- // is a field node's classifier no longer reads at all — so a run can come
170
- // back `failureKind:'auth'` (the server says login wall) while also
171
- // matching a vendor here (this CLI says Kasada/DataDome/etc), both true
172
- // signals about the same run, neither one wrong.
173
- //
174
- // trawl_cli#169 precedent: never assert client-side what only the server
175
- // knows. Silently picking a winner between two disagreeing signals is
176
- // exactly that — printing "no reliable bypass" alone asserts the run is
177
- // NOT an auth wall (may stop someone from trying a fixable cookie
178
- // problem); printing the cookie hint alone asserts the run is NOT a
179
- // vendor wall (may send someone chasing cookies against a wall no cookie
180
- // fixes). This CLI cannot tell which is true without either importing
181
- // node's BLOCK_TYPE_PATTERN (a second source of truth for the same list —
182
- // how this defect got here) or re-detecting the login URL itself (the
183
- // client-side re-derivation this fix deliberately rejects). So when both
184
- // fire, render ONE hedged line naming both readings instead of picking:
185
- // never emit the bare "no reliable bypass" verdict in that case — it is
186
- // exactly the false precision this branch exists to avoid.
187
- //
188
- // `conflictingSignals` is deliberately NOT gated on `scrapId` the way the
189
- // plain hint below is: the false "no reliable bypass" verdict is wrong
190
- // regardless of whether we also have an id to build an actionable command
191
- // for, so a scrapId-less conflicting run must still avoid it. Only the
192
- // action line degrades (to a generic Settings pointer) when there's no id.
193
- //
194
- // Single-signal cases are untouched: a plain vendor match (no
195
- // failureKind:'auth') still gets "no reliable bypass"; a plain
196
- // failureKind:'auth' (no vendor match) still gets the plain hint block
197
- // below. `errMsg` stays suppressed whenever any hint form (conflicting or
198
- // plain) is about to render — node's own `errorMessage` for an auth run
199
- // (historys.service.js ~line 774) IS the hint copy verbatim, literal
200
- // `<id>` placeholder and all, so printing it here would just be a second
201
- // rendering of the same information. If neither hint form renders (no
202
- // vendor conflict and no scrapId for the plain hint), fall back to node's
203
- // raw message rather than dropping the error entirely.
204
68
  const wallVendor = detectWallVendor(run);
205
69
  const errMsg = run.errorMessage ?? run.errorSnapshot?.errorMessage;
206
- // trawl_cli#182 — same `status`/`statusDetail` guard as `detectWallVendor`
207
- // above, and for the same reason (see its doc comment): `failureKind` is a
208
- // terminal classification trawl_node stamps once
209
- // (modules/historys/services/historys.service.js ~line 837), but
210
- // `patchForRegression` (modules/scraps/services/scraps.service.js ~line
211
- // 1972) can flip `status`/`statusDetail` to success/regression LATER,
212
- // without ever clearing `failureKind`. Left unguarded, a run that
213
- // ultimately succeeded or degraded could still carry a stale
214
- // `failureKind:'auth'` and render the login-wall hint (or the
215
- // conflicting-signals hedge) right next to a green "success" badge or a
216
- // "Regression: length N vs baseline M" line — the exact contradiction
217
- // `detectWallVendor`'s guard already exists to prevent for `block.kind`.
218
- // One rule, not two coincidences: both checks guard the same two fields
219
- // against the same after-the-fact patch.
220
70
  const authWall = isAuthWall(run);
221
71
  const authHintWillRender = authWall && Boolean(scrapId);
222
72
  const conflictingSignals = Boolean(wallVendor) && authWall;
@@ -232,69 +82,38 @@ export function formatDoctor(scrapTitle, run, fix = null, scrapId) {
232
82
  else if (errMsg && !authHintWillRender) {
233
83
  lines.push(chalk.dim(' Error: ') + chalk.red(errMsg));
234
84
  }
235
- // Failed selector
236
85
  if (run.errorSnapshot?.selector) {
237
86
  lines.push(chalk.dim(' Failed selector: ') + chalk.cyan(run.errorSnapshot.selector));
238
87
  }
239
- // Blocked
240
88
  lines.push(chalk.dim(' Blocked: ') + (run.blocked ? chalk.red('yes') : 'no'));
241
- // Proxy tier (abstract, kept for owner)
242
89
  if (run.proxyTier && TIER_LABELS[run.proxyTier]) {
243
90
  lines.push(chalk.dim(' Proxy: ') + TIER_LABELS[run.proxyTier]);
244
91
  }
245
- // Empty context detail
246
92
  if (run.statusDetail === 'empty' && run.emptyContext?.page) {
247
93
  const page = run.emptyContext.page;
248
94
  lines.push(chalk.dim(' Empty context: ') + `url=${page.url ?? '?'} anchors=${page.totalAnchors ?? '?'}`);
249
95
  }
250
- // Login-wall hint (trawl_node#1975 `failureKind==='auth'`) — trust node's
251
- // verdict, never re-detect the login URL client-side. Hypothesis-framed
252
- // ("likely fix"), not a promise: measured cases (Reddit/X/Instagram) had
253
- // valid cookies and stayed walled. Gated on `scrapId` alone (via
254
- // `authHintWillRender` above, shared with the generic error line so the
255
- // two can't drift apart), NOT the `scrapId ?? run._id` fallback used
256
- // elsewhere here — `run._id` is a history id, and the session-set command
257
- // on one silently targets the wrong document.
258
- //
259
- // trawl_cli#182 — excludes `conflictingSignals`: when a wall-vendor match
260
- // also fired, the combined hedged line above already covers the cookie
261
- // suggestion. Rendering this plain, unhedged block too would stack a
262
- // second verdict under the first (and directly contradict it, since this
263
- // block asserts the run IS a login wall with no caveat) — exactly the
264
- // two-verdicts-for-one-run defect this fix removes. Exactly one verdict
265
- // block renders per run.
266
96
  if (authHintWillRender && !conflictingSignals) {
267
97
  lines.push('');
268
98
  lines.push(chalk.yellow(' Login wall (hypothesis): ') + 'this looks like a login redirect — your own session cookies are the likely fix.');
269
99
  lines.push(chalk.dim(` → trawl scraps account session set ${scrapId} -c cookies.json (or app Settings → Account)`));
270
100
  }
271
- // Regression detail. trawl_node#1950 dropped the redundant `regressionDetected`
272
- // boolean — it was `true` on every row iff `statusDetail === 'regression'`,
273
- // so read that directly (works against a pre- or post-#1950 backend alike,
274
- // no CLI/backend sequencing gap: `statusDetail` isn't part of this rename).
275
101
  if (run.statusDetail === 'regression') {
276
102
  lines.push(chalk.dim(' Regression: ') + `length ${run.length ?? '?'} vs baseline ${run.baselineLength ?? '?'}`);
277
103
  }
278
- // Autofix summary (when fix exists)
279
104
  if (fix) {
280
105
  lines.push(` ${chalk.magenta('Autofix:')} ${fix.outcome ?? '?'}${fix.classification ? ` — ${fix.classification}` : ''}${fix.reason ? ` (${fix.reason})` : ''}`);
281
106
  lines.push(chalk.dim(` → full diff/dry-run/knowledge: trawl scraps autofix ${scrapId ?? run._id} (or doctor --autofix)`));
282
107
  }
283
- // Timestamp
284
108
  if (run.createdAt) {
285
109
  lines.push(chalk.dim(' Run at: ') + formatDate(run.createdAt));
286
110
  }
287
- // Snapshot hint
288
111
  if (run.errorSnapshot?.html || run.statusDetail === 'empty' || run.statusDetail === 'error') {
289
112
  lines.push('');
290
113
  lines.push(chalk.dim(` → trawl scraps snapshot ${scrapId ?? run._id} --error (download error-path HTML)`));
291
114
  }
292
115
  return lines.join('\n');
293
116
  }
294
- /**
295
- * Pure formatter — returns a human-readable autofix detail block.
296
- * Shows diff, dry-run results, knowledge consulted.
297
- */
298
117
  export function formatAutofix(fix) {
299
118
  const lines = [];
300
119
  lines.push(chalk.magenta.bold('Autofix attempt'));
@@ -324,17 +143,12 @@ export function formatAutofix(fix) {
324
143
  }
325
144
  return lines.join('\n');
326
145
  }
327
- /**
328
- * Fetch the latest run + its ai_fix_end activity for a scrap.
329
- * Returns null if the scrap has no runs yet.
330
- */
331
146
  export async function fetchRunAndFix(scrapId) {
332
147
  const scrap = await api.get(`/api/scraps/${scrapId}`);
333
148
  const hid = scrap.history?.[0]?._id;
334
149
  if (!hid)
335
150
  return null;
336
151
  const run = await api.get(`/api/historys/${hid}`);
337
- // api.get unwraps the { data: T } envelope — the response is the array directly
338
152
  const acts = await api.get(`/api/scraps/${scrapId}/activities?history=${hid}&limit=5`);
339
153
  const fixAct = Array.isArray(acts) ? acts.find((a) => a.type === 'ai_fix_end') : undefined;
340
154
  const fix = fixAct ? { ...fixAct.metadata, createdAt: fixAct.createdAt } : null;