@trawlme/cli 3.11.0 → 3.12.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +5 -2
- package/dist/commands/create.d.ts +0 -28
- package/dist/commands/create.js +0 -89
- package/dist/commands/doctor.d.ts +0 -79
- package/dist/commands/doctor.js +1 -187
- package/dist/commands/login.js +0 -67
- package/dist/commands/ping.d.ts +0 -15
- package/dist/commands/ping.js +0 -15
- package/dist/commands/scraps.d.ts +0 -120
- package/dist/commands/scraps.js +142 -656
- package/dist/commands/skills.js +0 -22
- package/dist/commands/spec.d.ts +0 -85
- package/dist/commands/spec.js +0 -67
- package/dist/commands/telemetry.js +0 -4
- package/dist/commands/token.js +0 -28
- package/dist/commands/upgrade.js +0 -22
- package/dist/commands/whoami.d.ts +0 -12
- package/dist/commands/whoami.js +0 -6
- package/dist/index.d.ts +0 -188
- package/dist/index.js +0 -349
- package/dist/lib/api.d.ts +0 -78
- package/dist/lib/api.js +1 -320
- package/dist/lib/cdp-pipe.d.ts +31 -0
- package/dist/lib/cdp-pipe.js +141 -0
- package/dist/lib/chrome-discovery.d.ts +1 -0
- package/dist/lib/chrome-discovery.js +30 -0
- package/dist/lib/chrome-launch.d.ts +8 -0
- package/dist/lib/chrome-launch.js +53 -0
- package/dist/lib/config.d.ts +0 -53
- package/dist/lib/config.js +0 -55
- package/dist/lib/confirm.d.ts +0 -55
- package/dist/lib/confirm.js +0 -47
- package/dist/lib/docs.d.ts +0 -123
- package/dist/lib/docs.js +0 -169
- package/dist/lib/errors.d.ts +0 -134
- package/dist/lib/errors.js +0 -151
- package/dist/lib/format.d.ts +0 -6
- package/dist/lib/format.js +0 -6
- package/dist/lib/json.d.ts +0 -35
- package/dist/lib/json.js +0 -48
- package/dist/lib/jwt.d.ts +0 -7
- package/dist/lib/jwt.js +0 -7
- package/dist/lib/pinch.d.ts +0 -53
- package/dist/lib/pinch.js +6 -112
- package/dist/lib/pinchAnimation.d.ts +0 -16
- package/dist/lib/pinchAnimation.js +8 -29
- package/dist/lib/posthog.d.ts +0 -9
- package/dist/lib/posthog.js +0 -23
- package/dist/lib/prompt.js +1 -20
- package/dist/lib/secure-transport.d.ts +1 -0
- package/dist/lib/secure-transport.js +15 -0
- package/dist/lib/session-capture-guard.d.ts +6 -0
- package/dist/lib/session-capture-guard.js +9 -0
- package/dist/lib/session-capture.d.ts +55 -0
- package/dist/lib/session-capture.js +319 -0
- package/dist/lib/skills.d.ts +0 -175
- package/dist/lib/skills.js +1 -216
- package/dist/lib/skillsNudge.d.ts +0 -17
- package/dist/lib/skillsNudge.js +0 -83
- package/dist/lib/spinner.d.ts +0 -39
- package/dist/lib/spinner.js +0 -40
- package/dist/lib/storage-state.d.ts +55 -0
- package/dist/lib/storage-state.js +96 -0
- package/dist/lib/tips.d.ts +0 -38
- package/dist/lib/tips.js +0 -77
- package/dist/lib/updateCheckWorker.js +0 -14
- package/dist/lib/updateNotifier.d.ts +0 -17
- package/dist/lib/updateNotifier.js +0 -53
- package/dist/lib/validate.d.ts +0 -8
- package/dist/lib/validate.js +0 -8
- package/dist/lib/version.d.ts +0 -12
- package/dist/lib/version.js +1 -13
- package/package.json +2 -2
package/README.md
CHANGED
|
@@ -36,7 +36,7 @@ Four methods:
|
|
|
36
36
|
3. **Token flag** — `trawl login --token <jwt>` (CI/CD, direct JWT)
|
|
37
37
|
4. **API key** — `TRAWL_API_KEY=trawl_xxx trawl list` (scoped, revocable one-by-one — the recommended credential for an agent driving this CLI; see [docs/agent-quickstart.md](docs/agent-quickstart.md))
|
|
38
38
|
|
|
39
|
-
A `TRAWL_API_KEY` is a `trawl_*`-prefixed credential (create/revoke one in the Trawl dashboard) sent as `Authorization: Bearer` instead of the session `Cookie: TOKEN=` a JWT uses — `trawl` picks the right one automatically based on the credential's own shape, never a flag. `TRAWL_API_KEY` wins over `TRAWL_TOKEN` when both happen to be set. It is scoped server-side — in practice to a set of scraps, since the dashboard's key-create form sets only the scrap allow-list; narrowing which *actions* a key may perform needs an explicit `scopes` array at creation over the API, and a key without one has every action granted. It works on most of the Core tier — `create`/`list`/`get`/`data`/`history`/`run-info`/`run`/`trigger`/`ping`, including the `--watch` **polling flag** on `run`/`trigger` — plus, outside Core, `scraps account status`/`scraps doctor`/`scraps autofix` (all three only ever read routes trawl_node opened to keys). It does **not** work on `whoami`, `scraps update`/`delete`, `scraps account set`/`delete`/`clear-session`/`session set`, `scraps banner`, `scraps snapshot` (the scrap lookup it starts from is dual-auth, but the `html-snapshot` route it downloads from isn't), or the standalone SSE **command** `scraps watch` (do not conflate the two: `--watch` is a flag on `run`/`trigger` and works under a key; `scraps watch` is a separate command and is JWT-only) — those stay JWT-only and fail with a `kind:"auth"` envelope (exit `3`) pointing at `trawl login` under a key. This list mirrors trawl_node's route wiring as of this writing, not a frozen guarantee — for anything not named here, trust the real `--json` envelope over this paragraph. Never send both a key and a JWT on the same request — the CLI only ever attaches one. `list --unhealthy` (and the plain health badge on `list`/`get`) rides the same `GET /api/scraps`/`GET /api/scraps/:id` routes as the rest of the group — no separate auth path — so it works under a key exactly like plain `list`/`get`, and a scrap-scoped key still only ever sees its own allow-listed scraps through the filter.
|
|
39
|
+
A `TRAWL_API_KEY` is a `trawl_*`-prefixed credential (create/revoke one in the Trawl dashboard) sent as `Authorization: Bearer` instead of the session `Cookie: TOKEN=` a JWT uses — `trawl` picks the right one automatically based on the credential's own shape, never a flag. `TRAWL_API_KEY` wins over `TRAWL_TOKEN` when both happen to be set. It is scoped server-side — in practice to a set of scraps, since the dashboard's key-create form sets only the scrap allow-list; narrowing which *actions* a key may perform needs an explicit `scopes` array at creation over the API, and a key without one has every action granted. It works on most of the Core tier — `create`/`list`/`get`/`data`/`history`/`run-info`/`run`/`trigger`/`ping`, including the `--watch` **polling flag** on `run`/`trigger` — plus, outside Core, `scraps account status`/`scraps doctor`/`scraps autofix` (all three only ever read routes trawl_node opened to keys). It does **not** work on `whoami`, `scraps update`/`delete`, `scraps account set`/`delete`/`clear-session`/`session set`/`session capture`, `scraps banner`, `scraps snapshot` (the scrap lookup it starts from is dual-auth, but the `html-snapshot` route it downloads from isn't), or the standalone SSE **command** `scraps watch` (do not conflate the two: `--watch` is a flag on `run`/`trigger` and works under a key; `scraps watch` is a separate command and is JWT-only) — those stay JWT-only and fail with a `kind:"auth"` envelope (exit `3`) pointing at `trawl login` under a key. This list mirrors trawl_node's route wiring as of this writing, not a frozen guarantee — for anything not named here, trust the real `--json` envelope over this paragraph. Never send both a key and a JWT on the same request — the CLI only ever attaches one. `list --unhealthy` (and the plain health badge on `list`/`get`) rides the same `GET /api/scraps`/`GET /api/scraps/:id` routes as the rest of the group — no separate auth path — so it works under a key exactly like plain `list`/`get`, and a scrap-scoped key still only ever sees its own allow-listed scraps through the filter.
|
|
40
40
|
|
|
41
41
|
Custom API URL: `trawl login --url https://self-hosted.example.com`
|
|
42
42
|
|
|
@@ -113,9 +113,12 @@ trawl scraps account delete <id> [--force] [--json]
|
|
|
113
113
|
trawl scraps account clear-session <id> [--json]
|
|
114
114
|
trawl scraps account status <id> [--json]
|
|
115
115
|
trawl scraps account session set <id> -c <file> [--json]
|
|
116
|
+
trawl scraps account session capture <id> [--chrome <path>] [--json]
|
|
116
117
|
```
|
|
117
118
|
|
|
118
|
-
`account session set` uploads a
|
|
119
|
+
`account session set` uploads a cookie JSON array — or a `{ cookies, origins }` storageState file (see `session capture` below) — to bootstrap a logged-in session without storing credentials (BYO-cookies). A missing `-u/--username`/`-p/--password` on `account set` follows the same [non-interactive rule](#non-interactive-rule) as `login`.
|
|
120
|
+
|
|
121
|
+
`account session capture <id>` is the one-command version of the same idea: it opens a real, visible Chrome window at the scrap's target URL, waits for **you** to log in there — 2FA included — then reads the resulting session over the Chrome DevTools Protocol (cookies **and** per-origin localStorage) and uploads it. You stay authenticated as yourself; Trawl never sees your credentials, and responsibility for lawful use of the captured session stays with you. The capture is scoped to the scrap's own target host by RFC 6265 domain-matching, the same rule a real browser uses to decide which cookies to send: a cookie explicitly scoped to a domain (e.g. `Domain=example.com`) is in scope for that host and its subdomains, while a host-only cookie (no `Domain` attribute) is in scope only for the exact host it was set on — never a sibling subdomain, even one sharing a parent domain, and never another tenant on the same multi-tenant hosting domain (e.g. a different `*.github.io` site). localStorage follows the same anchor: an origin is in scope only if it's the target host itself or a subdomain of it. The one deliberate gap: a host-only cookie set on a sibling host (e.g. `auth.example.com` while the scrap targets `app.example.com`) is excluded — a real browser would never send it to the target host either, so nothing replay-relevant is lost unless the scrap script itself later navigates to that sibling. Completion is explicit: press Enter in the terminal once you're signed in — the primary path, and the only one guaranteed cross-platform. Closing the Chrome window also completes the capture (cookies only, not localStorage) when Chrome itself stays running after its last window closes, which this CLI verified on macOS; on a platform where closing the last window quits Chrome entirely, the command reports that fact instead of a session, and pressing Enter is the reliable path there. This command needs a local interactive terminal with a real display; it does not work headless, in CI, or over a plain SSH session, and auto-detects a system Chrome/Chromium (override with `--chrome <path>` or `TRAWL_CHROME_PATH`). Never prints the captured session itself, in any mode — `--json` echoes only the server's response plus capture counts.
|
|
119
122
|
|
|
120
123
|
### Claude Code skills
|
|
121
124
|
|
|
@@ -1,37 +1,9 @@
|
|
|
1
1
|
import { Command } from 'commander';
|
|
2
|
-
/**
|
|
3
|
-
* `POST /api/ai/wizard` response contract (#114, trawl_node —
|
|
4
|
-
* ai.wizard.service.js#runWizard, shipped S1, contract LOCKED). The wizard
|
|
5
|
-
* chains, entirely server-side: AI code generation -> scrap creation ->
|
|
6
|
-
* the FIRST run (ScrapsService.load) -> auto-fix on failure (autoFix
|
|
7
|
-
* defaults true unless `--no-autofix` maps to `autoFix:false`).
|
|
8
|
-
*
|
|
9
|
-
* - `success` is an HONEST outcome of that first run, not "did the HTTP
|
|
10
|
-
* call succeed" — a failed first run is still a 200 response (the scrap
|
|
11
|
-
* itself was created either way; auto-fix, when enabled, retries in the
|
|
12
|
-
* background). Callers must branch on `success`, never assume 2xx means
|
|
13
|
-
* "the scrap works".
|
|
14
|
-
* - `scrap` is the full created Scrap object (present whenever creation got
|
|
15
|
-
* far enough to persist it — a hard failure before that point surfaces as
|
|
16
|
-
* a real HTTP error instead, handled by the shared error path).
|
|
17
|
-
* - `historyId` is best-effort (a failed server-side lookup leaves it
|
|
18
|
-
* `null`, never breaks the response) — it points at the first run just
|
|
19
|
-
* executed.
|
|
20
|
-
*/
|
|
21
2
|
export interface WizardResponse {
|
|
22
3
|
success: boolean;
|
|
23
4
|
scrap?: {
|
|
24
5
|
_id: string;
|
|
25
6
|
title: string;
|
|
26
|
-
/**
|
|
27
|
-
* Cron expression the wizard schedules this scrap on. As of #114/S1
|
|
28
|
-
* (trawl_node ai.wizard.service.js#runWizard's `scrapBody`) this is
|
|
29
|
-
* hardcoded server-side to a DAILY run — `'0 7 * * *'` / `cronTimezone:
|
|
30
|
-
* 'UTC'` — unconditionally, regardless of `--prompt`/`--no-autofix`.
|
|
31
|
-
* Present on the scrap object returned here (the wizard controller
|
|
32
|
-
* passes the created scrap straight through, no stripping) — change or
|
|
33
|
-
* disable it with `trawl scraps update <id> --cron <expr>` / `--no-cron`.
|
|
34
|
-
*/
|
|
35
7
|
cron?: string | null;
|
|
36
8
|
cronTimezone?: string;
|
|
37
9
|
[key: string]: unknown;
|
package/dist/commands/create.js
CHANGED
|
@@ -8,12 +8,6 @@ import { requireUrl, requireString } from '../lib/validate.js';
|
|
|
8
8
|
import { UsageError } from '../lib/errors.js';
|
|
9
9
|
import { renderPinch, pinchEnabled } from '../lib/pinch.js';
|
|
10
10
|
import { startPinchAnimation } from '../lib/pinchAnimation.js';
|
|
11
|
-
/** Best-effort, honest first-run summary — never claims a background retry
|
|
12
|
-
* happened when auto-fix was disabled for this call, and never claims a
|
|
13
|
-
* scrap was persisted when the response carries none (#114-F3 — a hard
|
|
14
|
-
* failure before persistence still comes back as `success:false` with no
|
|
15
|
-
* `scrap` at all; claiming "auto-fix retrying in the background" then would
|
|
16
|
-
* be fabricated — there is nothing to retry). */
|
|
17
11
|
function firstRunLabel(data, autoFixEnabled) {
|
|
18
12
|
if (data.success)
|
|
19
13
|
return 'succeeded';
|
|
@@ -23,10 +17,6 @@ function firstRunLabel(data, autoFixEnabled) {
|
|
|
23
17
|
return 'failed (auto-fix retrying in the background)';
|
|
24
18
|
return 'failed';
|
|
25
19
|
}
|
|
26
|
-
/** Best-effort human description of a daily cron (`M H * * *`) — the only
|
|
27
|
-
* shape the wizard's server-side default currently produces. Falls back to
|
|
28
|
-
* printing the raw expression for anything else rather than guessing at a
|
|
29
|
-
* schedule the CLI can't actually parse. */
|
|
30
20
|
function describeCron(cron) {
|
|
31
21
|
const match = /^(\d{1,2})\s+(\d{1,2})\s+\*\s+\*\s+\*$/.exec(cron.trim());
|
|
32
22
|
if (!match)
|
|
@@ -34,16 +24,6 @@ function describeCron(cron) {
|
|
|
34
24
|
const [, min, hour] = match;
|
|
35
25
|
return `daily ${hour.padStart(2, '0')}:${min.padStart(2, '0')}`;
|
|
36
26
|
}
|
|
37
|
-
/**
|
|
38
|
-
* #114-F2 — a wizard-created scrap runs on a DAILY cron by default
|
|
39
|
-
* server-side; surface that up front rather than leaving it to be
|
|
40
|
-
* discovered later as an unexpected recurring quota charge. Reads the real
|
|
41
|
-
* `cron`/`cronTimezone` field off the response scrap when present; falls
|
|
42
|
-
* back to the known wizard default wording ONLY when the field is missing
|
|
43
|
-
* from the response (an older server, or a future rename) — never invents a
|
|
44
|
-
* schedule value that might not match what the server actually applied.
|
|
45
|
-
* Returns null when there's no scrap to schedule at all.
|
|
46
|
-
*/
|
|
47
27
|
function scheduleLabel(scrap) {
|
|
48
28
|
if (!scrap)
|
|
49
29
|
return null;
|
|
@@ -53,24 +33,11 @@ function scheduleLabel(scrap) {
|
|
|
53
33
|
}
|
|
54
34
|
return 'scheduled daily by default';
|
|
55
35
|
}
|
|
56
|
-
/**
|
|
57
|
-
* #121 — onboarding principle: a successful `create` should SHOW the value,
|
|
58
|
-
* not just an id. Best-effort fetch of the first run's persisted data (the
|
|
59
|
-
* same read-only, no-quota path `trawl data <id>` uses:
|
|
60
|
-
* `GET /api/historys/:historyId` → a JSON string `{ data: [...] }`) and print
|
|
61
|
-
* a small proof-of-value sample (count + first-item keys + one truncated
|
|
62
|
-
* value line — never a full dump; that's what `trawl data --json` is for).
|
|
63
|
-
* ANY failure (network, parse, no data) is swallowed silently — the sample is
|
|
64
|
-
* a bonus, it must never turn a successful create into a failure or noise.
|
|
65
|
-
*/
|
|
66
36
|
async function printDataSample(historyId) {
|
|
67
37
|
try {
|
|
68
38
|
const detail = await api.get(`/api/historys/${historyId}`);
|
|
69
39
|
if (typeof detail?.data !== 'string' || !detail.data)
|
|
70
40
|
return;
|
|
71
|
-
// #159 — same JSON-string-within-JSON shape as scrap.history[0].data;
|
|
72
|
-
// tolerate the documented raw-control-char server quirk here too instead
|
|
73
|
-
// of silently losing the sample to the outer try/catch below.
|
|
74
41
|
const items = parseServerJson(detail.data)?.data;
|
|
75
42
|
if (!Array.isArray(items) || items.length === 0)
|
|
76
43
|
return;
|
|
@@ -88,7 +55,6 @@ async function printDataSample(historyId) {
|
|
|
88
55
|
}
|
|
89
56
|
}
|
|
90
57
|
catch {
|
|
91
|
-
// best-effort — a missing sample never fails or noises up a good create
|
|
92
58
|
}
|
|
93
59
|
}
|
|
94
60
|
export const create = new Command('create')
|
|
@@ -99,15 +65,6 @@ export const create = new Command('create')
|
|
|
99
65
|
.option('--no-autofix', 'Disable AI auto-fix on first-run failure (default: on)')
|
|
100
66
|
.option('--json', 'Output the raw API payload')
|
|
101
67
|
.action(async (rawUrl, opts) => {
|
|
102
|
-
// Fast, local usage-errors (exit 2) — never a round-trip to the server
|
|
103
|
-
// for something we can already tell is bad. Same fail-fast pattern the
|
|
104
|
-
// former `fetch` command used for its URL argument.
|
|
105
|
-
//
|
|
106
|
-
// #116 — the url is accepted BOTH as the positional argument (agent
|
|
107
|
-
// one-liner) and as `--url` (muscle memory from `scraps create` and the
|
|
108
|
-
// API body {url, goal}). Exactly one is required; both are fine only
|
|
109
|
-
// when identical — two DIFFERENT urls is ambiguous, refuse loudly
|
|
110
|
-
// rather than silently picking one.
|
|
111
68
|
if (rawUrl === undefined && opts.url === undefined) {
|
|
112
69
|
throw new UsageError("missing required argument 'url' (positional, or --url <url>)");
|
|
113
70
|
}
|
|
@@ -121,58 +78,28 @@ export const create = new Command('create')
|
|
|
121
78
|
goal,
|
|
122
79
|
...(opts.autofix === false && { autoFix: false }),
|
|
123
80
|
};
|
|
124
|
-
// #91/#106-F1 long-run pattern — the wizard runs AI generation + scrap
|
|
125
|
-
// creation + a real FIRST run (+ autofix retries) entirely server-side,
|
|
126
|
-
// legitimately 30-250s+. The 30s DEFAULT_TIMEOUT_MS would abort it
|
|
127
|
-
// mid-flight and fabricate a NetworkError timeout for a request that was
|
|
128
|
-
// always going to succeed.
|
|
129
81
|
const call = () => api.post('/api/ai/wizard', body, { timeoutMs: LONG_RUN_TIMEOUT_MS });
|
|
130
|
-
// #106-F2 pattern carried over from `fetch` — under --json stdout must
|
|
131
|
-
// be provably pure: no spinner channel at all. Only the human path gets
|
|
132
|
-
// the ora progress indicator; --json calls the API directly.
|
|
133
82
|
let data;
|
|
134
83
|
if (opts.json) {
|
|
135
84
|
data = await call();
|
|
136
85
|
}
|
|
137
86
|
else {
|
|
138
|
-
// #122/#131 — while the server-side wizard runs (legitimately 30-250s+,
|
|
139
|
-
// see LONG_RUN_TIMEOUT_MS above) Pinch ANIMATES in place (claw-wiggle)
|
|
140
|
-
// on a color-capable TTY; otherwise the ora spinner (which itself
|
|
141
|
-
// no-ops under a non-TTY, so piped human output stays clean). Both
|
|
142
|
-
// write to stderr only — stdout purity under `--json` is guaranteed by
|
|
143
|
-
// the `opts.json` branch above, never by this code.
|
|
144
87
|
const anim = pinchEnabled() ? startPinchAnimation(`Creating a scrap from ${url}…`) : null;
|
|
145
88
|
try {
|
|
146
89
|
data = anim
|
|
147
90
|
? await call()
|
|
148
91
|
: await spin(call, {
|
|
149
|
-
// No verdict symbol (#106-F3) — ora's success only means "the
|
|
150
|
-
// HTTP call didn't throw", not "the first run succeeded".
|
|
151
92
|
text: `Creating a scrap from ${url}…`,
|
|
152
93
|
successText: 'Request complete',
|
|
153
94
|
});
|
|
154
95
|
}
|
|
155
96
|
catch (err) {
|
|
156
97
|
anim?.stop();
|
|
157
|
-
// #114-F1 — a client-side timeout (NetworkError, "timed out after
|
|
158
|
-
// …ms" per api.ts's safeFetch) does NOT mean the wizard failed
|
|
159
|
-
// server-side: the scrap creation + first run keep going on the
|
|
160
|
-
// server after the CLI gives up waiting, so the scrap may already
|
|
161
|
-
// exist (or land moments later). Warn BEFORE rethrowing so a retry
|
|
162
|
-
// isn't the first instinct — a blind retry creates a duplicate scrap
|
|
163
|
-
// and burns AI-generation quota a second time for the same goal.
|
|
164
|
-
// Only a NetworkError whose message identifies it as the timeout
|
|
165
|
-
// branch qualifies — a DNS/connection-refused NetworkError never
|
|
166
|
-
// reached the server at all, so there's nothing to warn about here.
|
|
167
98
|
if (err instanceof NetworkError && /timed out/i.test(err.message)) {
|
|
168
99
|
console.error(chalk.yellow('⚠ The request timed out client-side, but the scrap may STILL have been created server-side — run `trawl list` before retrying (a retry creates a DUPLICATE scrap + burns quota).'));
|
|
169
100
|
}
|
|
170
|
-
// Rethrow unchanged so index.ts's classifyError/exit-code taxonomy
|
|
171
|
-
// stays intact (this stays a NetworkError -> exit 5, same as before).
|
|
172
101
|
throw err;
|
|
173
102
|
}
|
|
174
|
-
// Success — stop + erase the animation so the result prints on a clean
|
|
175
|
-
// line (the celebrating/confused frame below is the one-shot outcome).
|
|
176
103
|
anim?.stop();
|
|
177
104
|
}
|
|
178
105
|
if (opts.json) {
|
|
@@ -190,19 +117,11 @@ export const create = new Command('create')
|
|
|
190
117
|
console.log(chalk.dim(` First run: `) + firstRunLabel(data, autoFixEnabled));
|
|
191
118
|
if (schedule)
|
|
192
119
|
console.log(chalk.dim(` Schedule: `) + schedule);
|
|
193
|
-
// #121 — on a successful first run, prove the value: show a small data
|
|
194
|
-
// sample (best-effort, silent on failure) before the next-step hint.
|
|
195
120
|
if (data.success && data.historyId)
|
|
196
121
|
await printDataSample(data.historyId);
|
|
197
|
-
// #121 — the natural next command is the DATA, not the metadata: point
|
|
198
|
-
// at `trawl data` primarily, keep `trawl get` as the secondary detail view.
|
|
199
122
|
if (scrapId) {
|
|
200
123
|
console.log(chalk.dim(` Next step: `) + `trawl data ${scrapId}` + chalk.dim(` (details: trawl get ${scrapId})`));
|
|
201
124
|
}
|
|
202
|
-
// #114-F3 — only claim a background retry is happening when a scrap
|
|
203
|
-
// actually exists to retry (never fabricate progress that isn't real);
|
|
204
|
-
// `firstRunLabel` above already covers the !scrap / autofix-disabled
|
|
205
|
-
// wording, this adds the actionable poll target on top.
|
|
206
125
|
if (!data.success && data.scrap && autoFixEnabled) {
|
|
207
126
|
const pollTarget = data.historyId
|
|
208
127
|
? `\`trawl data ${scrapId}\` or \`trawl run-info ${data.historyId}\``
|
|
@@ -210,18 +129,10 @@ export const create = new Command('create')
|
|
|
210
129
|
console.log(chalk.yellow(` Note: `) +
|
|
211
130
|
`Auto-fix is retrying in the background — do NOT re-run create; poll ${pollTarget}.`);
|
|
212
131
|
}
|
|
213
|
-
// #122 — Pinch reacts to the HONEST first-run outcome (same signal the
|
|
214
|
-
// ✓/✗ line above already renders): celebrates a real success, looks
|
|
215
|
-
// confused on a genuine failure. Belt-and-suspenders `!opts.json`
|
|
216
|
-
// alongside pinchEnabled() — see the 'thinking' print above for why.
|
|
217
132
|
if (!opts.json && pinchEnabled()) {
|
|
218
133
|
console.log(data.success ? renderPinch('celebrating') : renderPinch('confused'));
|
|
219
134
|
}
|
|
220
135
|
}
|
|
221
|
-
// Honest exit code alongside the honest payload — a --json caller gets
|
|
222
|
-
// the raw body regardless (never wrapped/altered), but a script checking
|
|
223
|
-
// the exit code alone must be able to tell "first run failed" from "ran
|
|
224
|
-
// fine" without parsing. Same contract the former `fetch` command used.
|
|
225
136
|
if (!data.success)
|
|
226
137
|
process.exitCode = 1;
|
|
227
138
|
});
|
|
@@ -1,9 +1,3 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Run diagnostics shape — mirrors the owner-safe REST shape returned by
|
|
3
|
-
* GET /api/historys/:id (Phase 1 projection, 2026-05-27).
|
|
4
|
-
* Cost and proxy-nature fields are intentionally absent by design.
|
|
5
|
-
* Abstract proxyTier (Tier 0–4) is kept.
|
|
6
|
-
*/
|
|
7
1
|
export interface Run {
|
|
8
2
|
_id: string;
|
|
9
3
|
status: boolean | null;
|
|
@@ -28,11 +22,6 @@ export interface Run {
|
|
|
28
22
|
block?: {
|
|
29
23
|
kind?: string | null;
|
|
30
24
|
} | null;
|
|
31
|
-
/** @deprecated trawl_node#1950 renamed this to `block.kind`. Kept as a read
|
|
32
|
-
* fallback (see detectWallVendor) for the window where this CLI is
|
|
33
|
-
* published ahead of the trawl_node prod tag — a prod backend served from
|
|
34
|
-
* the pre-#1950 tag still returns this flat field, not `block.kind`. Drop
|
|
35
|
-
* once prod is confirmed on a tag containing #1950. */
|
|
36
25
|
blockType?: string | null;
|
|
37
26
|
proxyTier?: string | null;
|
|
38
27
|
baselineLength?: number | null;
|
|
@@ -40,64 +29,14 @@ export interface Run {
|
|
|
40
29
|
createdAt?: string;
|
|
41
30
|
time?: number | null;
|
|
42
31
|
triggeredBy?: string | null;
|
|
43
|
-
/** trawl_node#1975 — freeform failure classification (`'auth'` = login-wall
|
|
44
|
-
* empty run, cookies are the fix). Not a TS union — trawl_node's own set
|
|
45
|
-
* is additive/open, so this CLI must stay read-safe against a future
|
|
46
|
-
* value it doesn't know about yet, same posture as `block.kind` above. */
|
|
47
32
|
failureKind?: string | null;
|
|
48
33
|
}
|
|
49
|
-
/**
|
|
50
|
-
* Resolve a known anti-bot vendor name from a worker `block.kind`
|
|
51
|
-
* string. Returns null when the run succeeded, when there is no `block.kind`
|
|
52
|
-
* signal at all, or when `block.kind` names something other than a known
|
|
53
|
-
* vendor (e.g. `proxy-domain-gate`, `rate_limited_per_host`) — those stay on
|
|
54
|
-
* the genuine-error path since we can't honestly attribute them to a specific
|
|
55
|
-
* "no reliable bypass" wall.
|
|
56
|
-
*
|
|
57
|
-
* `blocked` is deliberately NOT consulted here (#177): it is a mid-run,
|
|
58
|
-
* attempt-level signal the worker only stamps on one envelope shape, so it
|
|
59
|
-
* reads `false` on the great majority of genuinely walled runs (the early-block
|
|
60
|
-
* throw path sets `block.kind` but never `blocked` — 63 of 64 walled runs
|
|
61
|
-
* measured on prod). `status === true` wins instead: a run that ultimately
|
|
62
|
-
* returned data was not walled, even if an earlier tier's `block.kind` stamp
|
|
63
|
-
* survived on the row.
|
|
64
|
-
*
|
|
65
|
-
* trawl_node#1950 renamed the flat `blockType` field to nested `block.kind`.
|
|
66
|
-
* Read `block?.kind` first, falling back to the deprecated flat `blockType` —
|
|
67
|
-
* this CLI can be published (and talk to prod) before the trawl_node prod tag
|
|
68
|
-
* containing #1950 is cut (see the `Run.blockType` doc comment), so a prod
|
|
69
|
-
* response can still be the pre-#1950 flat shape for a while.
|
|
70
|
-
*/
|
|
71
34
|
export declare function detectWallVendor(run: Pick<Run, 'status' | 'statusDetail' | 'block' | 'blockType'>): string | null;
|
|
72
|
-
/**
|
|
73
|
-
* #184 — true for a run currently carrying a live (non-stale) login-wall
|
|
74
|
-
* verdict. Extracted out of `formatDoctor`'s own local `authWall` const so
|
|
75
|
-
* `commands/scraps.ts`'s `doctor`/`run-info` actions can reuse the EXACT
|
|
76
|
-
* same guard to decide whether to fire the skills-install safety-net nudge
|
|
77
|
-
* (lib/skillsNudge.ts `maybeSuggestSkillsForAuthWall`) — one rule, not two
|
|
78
|
-
* copies that could quietly drift apart.
|
|
79
|
-
*
|
|
80
|
-
* Same staleness guard as `detectWallVendor` above and for the same reason:
|
|
81
|
-
* `failureKind` is a terminal classification trawl_node stamps once, but
|
|
82
|
-
* `patchForRegression` can flip `status`/`statusDetail` to success/
|
|
83
|
-
* regression LATER without ever clearing it — so a run that ultimately
|
|
84
|
-
* succeeded or degraded must never still read as an active auth wall.
|
|
85
|
-
*
|
|
86
|
-
* Typed structurally loose (not `Pick<Run, ...>`) on purpose: `Run.status`/
|
|
87
|
-
* `statusDetail` are required fields, but `scraps.ts`'s own `HistoryRun`
|
|
88
|
-
* (the run-info command's shape) declares the same three fields OPTIONAL —
|
|
89
|
-
* a `Pick<Run, ...>` parameter type would reject that caller at compile
|
|
90
|
-
* time even though every field it actually reads is present at runtime.
|
|
91
|
-
*/
|
|
92
35
|
export declare function isAuthWall(run: {
|
|
93
36
|
failureKind?: string | null;
|
|
94
37
|
status?: boolean | null;
|
|
95
38
|
statusDetail?: string | null;
|
|
96
39
|
}): boolean;
|
|
97
|
-
/**
|
|
98
|
-
* Autofix activity metadata — from the persisted ai_fix_end activity.
|
|
99
|
-
* aiUsage (cost) is stripped server-side; all diagnostics are kept.
|
|
100
|
-
*/
|
|
101
40
|
export interface FixActivity {
|
|
102
41
|
outcome: 'applied' | 'failed' | 'skipped' | 'breaker_tripped' | 'timeout' | null;
|
|
103
42
|
classification?: string | null;
|
|
@@ -127,26 +66,8 @@ interface ScrapHead {
|
|
|
127
66
|
}
|
|
128
67
|
export declare function pickRun(run: Run): Partial<Run>;
|
|
129
68
|
export declare function pickFix(fix: FixActivity | null): Partial<FixActivity> | null;
|
|
130
|
-
/**
|
|
131
|
-
* Pure formatter — returns a human-readable diagnosis string for a run.
|
|
132
|
-
* Used by `doctor`, `doctor --autofix`, and `data --errors`.
|
|
133
|
-
*
|
|
134
|
-
* @param scrapTitle - Display name of the scrap
|
|
135
|
-
* @param run - Owner-safe history run object
|
|
136
|
-
* @param fix - Optional autofix activity (null = no fix attempted)
|
|
137
|
-
* @param scrapId - Scrap document id (used in hint lines — autofix/snapshot take scrap id, not run id)
|
|
138
|
-
* @returns Multi-line string ready for console.log
|
|
139
|
-
*/
|
|
140
69
|
export declare function formatDoctor(scrapTitle: string, run: Run, fix?: FixActivity | null, scrapId?: string): string;
|
|
141
|
-
/**
|
|
142
|
-
* Pure formatter — returns a human-readable autofix detail block.
|
|
143
|
-
* Shows diff, dry-run results, knowledge consulted.
|
|
144
|
-
*/
|
|
145
70
|
export declare function formatAutofix(fix: FixActivity): string;
|
|
146
|
-
/**
|
|
147
|
-
* Fetch the latest run + its ai_fix_end activity for a scrap.
|
|
148
|
-
* Returns null if the scrap has no runs yet.
|
|
149
|
-
*/
|
|
150
71
|
export declare function fetchRunAndFix(scrapId: string): Promise<{
|
|
151
72
|
scrap: ScrapHead;
|
|
152
73
|
run: Run;
|
package/dist/commands/doctor.js
CHANGED
|
@@ -1,27 +1,6 @@
|
|
|
1
1
|
import { api } from '../lib/api.js';
|
|
2
2
|
import chalk from 'chalk';
|
|
3
3
|
import { formatDate } from '../lib/format.js';
|
|
4
|
-
/**
|
|
5
|
-
* Known anti-bot vendors that the worker's `block.kind` field may name
|
|
6
|
-
* (ex-flat `blockType`, trawl_node#1950). Ordered by first-match;
|
|
7
|
-
* `block.kind` is a freeform worker string, not an enum (see the `Run.block`
|
|
8
|
-
* doc comment), so this is a best-effort substring match against the real
|
|
9
|
-
* field — never an invented/mocked value.
|
|
10
|
-
*
|
|
11
|
-
* `datadome`/`perimeterx`/`akamai`/`cloudflare` are literal substrings the
|
|
12
|
-
* current worker emits (detect.js). `kasada` is forward-compatible: the issue
|
|
13
|
-
* (#62 / wall-registry WS-7) names it as a genuine wall, but the current
|
|
14
|
-
* worker vocabulary does not yet emit it — this branch lights up
|
|
15
|
-
* automatically if/when a future worker classifier does, without a CLI
|
|
16
|
-
* change.
|
|
17
|
-
*
|
|
18
|
-
* trawl_cli#182 — an `auth` entry used to live here, rendering a login wall
|
|
19
|
-
* as "walled: auth — no reliable bypass". Removed: unlike a real anti-bot
|
|
20
|
-
* vendor, a login wall's own reliable bypass is the user's session cookies,
|
|
21
|
-
* so that verdict was wrong, not just early. `failureKind==='auth'`
|
|
22
|
-
* (trawl_node#1975) is the honest signal for a login wall — see the
|
|
23
|
-
* dedicated branch in `formatDoctor` below.
|
|
24
|
-
*/
|
|
25
4
|
const WALL_VENDOR_PATTERNS = [
|
|
26
5
|
[/datadome/i, 'DataDome'],
|
|
27
6
|
[/kasada/i, 'Kasada'],
|
|
@@ -29,66 +8,18 @@ const WALL_VENDOR_PATTERNS = [
|
|
|
29
8
|
[/akamai/i, 'Akamai'],
|
|
30
9
|
[/cloudflare/i, 'Cloudflare'],
|
|
31
10
|
];
|
|
32
|
-
/**
|
|
33
|
-
* Resolve a known anti-bot vendor name from a worker `block.kind`
|
|
34
|
-
* string. Returns null when the run succeeded, when there is no `block.kind`
|
|
35
|
-
* signal at all, or when `block.kind` names something other than a known
|
|
36
|
-
* vendor (e.g. `proxy-domain-gate`, `rate_limited_per_host`) — those stay on
|
|
37
|
-
* the genuine-error path since we can't honestly attribute them to a specific
|
|
38
|
-
* "no reliable bypass" wall.
|
|
39
|
-
*
|
|
40
|
-
* `blocked` is deliberately NOT consulted here (#177): it is a mid-run,
|
|
41
|
-
* attempt-level signal the worker only stamps on one envelope shape, so it
|
|
42
|
-
* reads `false` on the great majority of genuinely walled runs (the early-block
|
|
43
|
-
* throw path sets `block.kind` but never `blocked` — 63 of 64 walled runs
|
|
44
|
-
* measured on prod). `status === true` wins instead: a run that ultimately
|
|
45
|
-
* returned data was not walled, even if an earlier tier's `block.kind` stamp
|
|
46
|
-
* survived on the row.
|
|
47
|
-
*
|
|
48
|
-
* trawl_node#1950 renamed the flat `blockType` field to nested `block.kind`.
|
|
49
|
-
* Read `block?.kind` first, falling back to the deprecated flat `blockType` —
|
|
50
|
-
* this CLI can be published (and talk to prod) before the trawl_node prod tag
|
|
51
|
-
* containing #1950 is cut (see the `Run.blockType` doc comment), so a prod
|
|
52
|
-
* response can still be the pre-#1950 flat shape for a while.
|
|
53
|
-
*/
|
|
54
11
|
export function detectWallVendor(run) {
|
|
55
|
-
// A run that returned data was not walled. `status === true` is not the whole
|
|
56
|
-
// predicate: trawl_node flips a degraded-but-non-empty run to
|
|
57
|
-
// `status:false, statusDetail:'regression'` (scraps.service.js, #1112) via a
|
|
58
|
-
// patch that never clears `block.kind` — so keying on `status` alone would
|
|
59
|
-
// print "no reliable bypass" next to "Regression: length N vs baseline M",
|
|
60
|
-
// which is the same contradiction this guard exists to remove.
|
|
61
12
|
if (run.status === true || run.statusDetail === 'regression')
|
|
62
13
|
return null;
|
|
63
14
|
const kind = run.block?.kind ?? run.blockType ?? null;
|
|
64
15
|
if (!kind)
|
|
65
|
-
return null;
|
|
16
|
+
return null;
|
|
66
17
|
for (const [pattern, label] of WALL_VENDOR_PATTERNS) {
|
|
67
18
|
if (pattern.test(kind))
|
|
68
19
|
return label;
|
|
69
20
|
}
|
|
70
21
|
return null;
|
|
71
22
|
}
|
|
72
|
-
/**
|
|
73
|
-
* #184 — true for a run currently carrying a live (non-stale) login-wall
|
|
74
|
-
* verdict. Extracted out of `formatDoctor`'s own local `authWall` const so
|
|
75
|
-
* `commands/scraps.ts`'s `doctor`/`run-info` actions can reuse the EXACT
|
|
76
|
-
* same guard to decide whether to fire the skills-install safety-net nudge
|
|
77
|
-
* (lib/skillsNudge.ts `maybeSuggestSkillsForAuthWall`) — one rule, not two
|
|
78
|
-
* copies that could quietly drift apart.
|
|
79
|
-
*
|
|
80
|
-
* Same staleness guard as `detectWallVendor` above and for the same reason:
|
|
81
|
-
* `failureKind` is a terminal classification trawl_node stamps once, but
|
|
82
|
-
* `patchForRegression` can flip `status`/`statusDetail` to success/
|
|
83
|
-
* regression LATER without ever clearing it — so a run that ultimately
|
|
84
|
-
* succeeded or degraded must never still read as an active auth wall.
|
|
85
|
-
*
|
|
86
|
-
* Typed structurally loose (not `Pick<Run, ...>`) on purpose: `Run.status`/
|
|
87
|
-
* `statusDetail` are required fields, but `scraps.ts`'s own `HistoryRun`
|
|
88
|
-
* (the run-info command's shape) declares the same three fields OPTIONAL —
|
|
89
|
-
* a `Pick<Run, ...>` parameter type would reject that caller at compile
|
|
90
|
-
* time even though every field it actually reads is present at runtime.
|
|
91
|
-
*/
|
|
92
23
|
export function isAuthWall(run) {
|
|
93
24
|
return run.failureKind === 'auth' && run.status !== true && run.statusDetail !== 'regression';
|
|
94
25
|
}
|
|
@@ -120,28 +51,8 @@ export function pickFix(fix) {
|
|
|
120
51
|
.filter((k) => k in fix)
|
|
121
52
|
.map((k) => [k, fix[k]]));
|
|
122
53
|
}
|
|
123
|
-
/**
|
|
124
|
-
* Pure formatter — returns a human-readable diagnosis string for a run.
|
|
125
|
-
* Used by `doctor`, `doctor --autofix`, and `data --errors`.
|
|
126
|
-
*
|
|
127
|
-
* @param scrapTitle - Display name of the scrap
|
|
128
|
-
* @param run - Owner-safe history run object
|
|
129
|
-
* @param fix - Optional autofix activity (null = no fix attempted)
|
|
130
|
-
* @param scrapId - Scrap document id (used in hint lines — autofix/snapshot take scrap id, not run id)
|
|
131
|
-
* @returns Multi-line string ready for console.log
|
|
132
|
-
*/
|
|
133
54
|
export function formatDoctor(scrapTitle, run, fix = null, scrapId) {
|
|
134
55
|
const lines = [];
|
|
135
|
-
// Header + status badge. status:null means a run is IN FLIGHT (node
|
|
136
|
-
// persists {status:null, statusDetail:null, inFlight:true} the moment a
|
|
137
|
-
// run starts) — that must never render as "failed". (#88 item 1)
|
|
138
|
-
//
|
|
139
|
-
// #91 LOW — a regression row (status:false, statusDetail:'regression') was
|
|
140
|
-
// falling into the same red "failed" bucket as a genuine failure, then
|
|
141
|
-
// showing a contradictory "Regression: X vs Y" detail line right below it.
|
|
142
|
-
// A regression run's write actually SUCCEEDED (the item count just dropped
|
|
143
|
-
// vs baseline, detected by an async patch afterward — see #88 item 2 in
|
|
144
|
-
// scraps.ts) — it needs its own distinct badge, not "failed".
|
|
145
56
|
const ok = run.status === true;
|
|
146
57
|
const badge = ok
|
|
147
58
|
? chalk.green('● success')
|
|
@@ -154,69 +65,8 @@ export function formatDoctor(scrapTitle, run, fix = null, scrapId) {
|
|
|
154
65
|
: chalk.red('● failed');
|
|
155
66
|
lines.push(`${chalk.bold(scrapTitle)} ${badge}${run.statusDetail ? ` (${run.statusDetail})` : ''}`);
|
|
156
67
|
lines.push(chalk.dim(` Run ID: ${run._id}`));
|
|
157
|
-
// Error message — an honest accept-wall string for known-walled scraps
|
|
158
|
-
// (DataDome/Kasada/PerimeterX/Akamai terminal verdict from the worker),
|
|
159
|
-
// otherwise the real error (genuine transient failure).
|
|
160
|
-
//
|
|
161
|
-
// trawl_cli#182 — the wall-vendor verdict (this CLI's own
|
|
162
|
-
// WALL_VENDOR_PATTERNS matched against block.kind/blockType) and the
|
|
163
|
-
// login-wall hint (node's failureKind==='auth', trawl_node#1975) are two
|
|
164
|
-
// INDEPENDENT server-derived signals with no shared contract: node's block
|
|
165
|
-
// classifier (modules/historys/helpers/failureKind.js BLOCK_TYPE_PATTERN)
|
|
166
|
-
// and this CLI's vendor table live in different repos and can legitimately
|
|
167
|
-
// disagree. Concretely, `kasada` matches this CLI's table but is absent
|
|
168
|
-
// from node's block pattern, and the deprecated flat `blockType` fallback
|
|
169
|
-
// is a field node's classifier no longer reads at all — so a run can come
|
|
170
|
-
// back `failureKind:'auth'` (the server says login wall) while also
|
|
171
|
-
// matching a vendor here (this CLI says Kasada/DataDome/etc), both true
|
|
172
|
-
// signals about the same run, neither one wrong.
|
|
173
|
-
//
|
|
174
|
-
// trawl_cli#169 precedent: never assert client-side what only the server
|
|
175
|
-
// knows. Silently picking a winner between two disagreeing signals is
|
|
176
|
-
// exactly that — printing "no reliable bypass" alone asserts the run is
|
|
177
|
-
// NOT an auth wall (may stop someone from trying a fixable cookie
|
|
178
|
-
// problem); printing the cookie hint alone asserts the run is NOT a
|
|
179
|
-
// vendor wall (may send someone chasing cookies against a wall no cookie
|
|
180
|
-
// fixes). This CLI cannot tell which is true without either importing
|
|
181
|
-
// node's BLOCK_TYPE_PATTERN (a second source of truth for the same list —
|
|
182
|
-
// how this defect got here) or re-detecting the login URL itself (the
|
|
183
|
-
// client-side re-derivation this fix deliberately rejects). So when both
|
|
184
|
-
// fire, render ONE hedged line naming both readings instead of picking:
|
|
185
|
-
// never emit the bare "no reliable bypass" verdict in that case — it is
|
|
186
|
-
// exactly the false precision this branch exists to avoid.
|
|
187
|
-
//
|
|
188
|
-
// `conflictingSignals` is deliberately NOT gated on `scrapId` the way the
|
|
189
|
-
// plain hint below is: the false "no reliable bypass" verdict is wrong
|
|
190
|
-
// regardless of whether we also have an id to build an actionable command
|
|
191
|
-
// for, so a scrapId-less conflicting run must still avoid it. Only the
|
|
192
|
-
// action line degrades (to a generic Settings pointer) when there's no id.
|
|
193
|
-
//
|
|
194
|
-
// Single-signal cases are untouched: a plain vendor match (no
|
|
195
|
-
// failureKind:'auth') still gets "no reliable bypass"; a plain
|
|
196
|
-
// failureKind:'auth' (no vendor match) still gets the plain hint block
|
|
197
|
-
// below. `errMsg` stays suppressed whenever any hint form (conflicting or
|
|
198
|
-
// plain) is about to render — node's own `errorMessage` for an auth run
|
|
199
|
-
// (historys.service.js ~line 774) IS the hint copy verbatim, literal
|
|
200
|
-
// `<id>` placeholder and all, so printing it here would just be a second
|
|
201
|
-
// rendering of the same information. If neither hint form renders (no
|
|
202
|
-
// vendor conflict and no scrapId for the plain hint), fall back to node's
|
|
203
|
-
// raw message rather than dropping the error entirely.
|
|
204
68
|
const wallVendor = detectWallVendor(run);
|
|
205
69
|
const errMsg = run.errorMessage ?? run.errorSnapshot?.errorMessage;
|
|
206
|
-
// trawl_cli#182 — same `status`/`statusDetail` guard as `detectWallVendor`
|
|
207
|
-
// above, and for the same reason (see its doc comment): `failureKind` is a
|
|
208
|
-
// terminal classification trawl_node stamps once
|
|
209
|
-
// (modules/historys/services/historys.service.js ~line 837), but
|
|
210
|
-
// `patchForRegression` (modules/scraps/services/scraps.service.js ~line
|
|
211
|
-
// 1972) can flip `status`/`statusDetail` to success/regression LATER,
|
|
212
|
-
// without ever clearing `failureKind`. Left unguarded, a run that
|
|
213
|
-
// ultimately succeeded or degraded could still carry a stale
|
|
214
|
-
// `failureKind:'auth'` and render the login-wall hint (or the
|
|
215
|
-
// conflicting-signals hedge) right next to a green "success" badge or a
|
|
216
|
-
// "Regression: length N vs baseline M" line — the exact contradiction
|
|
217
|
-
// `detectWallVendor`'s guard already exists to prevent for `block.kind`.
|
|
218
|
-
// One rule, not two coincidences: both checks guard the same two fields
|
|
219
|
-
// against the same after-the-fact patch.
|
|
220
70
|
const authWall = isAuthWall(run);
|
|
221
71
|
const authHintWillRender = authWall && Boolean(scrapId);
|
|
222
72
|
const conflictingSignals = Boolean(wallVendor) && authWall;
|
|
@@ -232,69 +82,38 @@ export function formatDoctor(scrapTitle, run, fix = null, scrapId) {
|
|
|
232
82
|
else if (errMsg && !authHintWillRender) {
|
|
233
83
|
lines.push(chalk.dim(' Error: ') + chalk.red(errMsg));
|
|
234
84
|
}
|
|
235
|
-
// Failed selector
|
|
236
85
|
if (run.errorSnapshot?.selector) {
|
|
237
86
|
lines.push(chalk.dim(' Failed selector: ') + chalk.cyan(run.errorSnapshot.selector));
|
|
238
87
|
}
|
|
239
|
-
// Blocked
|
|
240
88
|
lines.push(chalk.dim(' Blocked: ') + (run.blocked ? chalk.red('yes') : 'no'));
|
|
241
|
-
// Proxy tier (abstract, kept for owner)
|
|
242
89
|
if (run.proxyTier && TIER_LABELS[run.proxyTier]) {
|
|
243
90
|
lines.push(chalk.dim(' Proxy: ') + TIER_LABELS[run.proxyTier]);
|
|
244
91
|
}
|
|
245
|
-
// Empty context detail
|
|
246
92
|
if (run.statusDetail === 'empty' && run.emptyContext?.page) {
|
|
247
93
|
const page = run.emptyContext.page;
|
|
248
94
|
lines.push(chalk.dim(' Empty context: ') + `url=${page.url ?? '?'} anchors=${page.totalAnchors ?? '?'}`);
|
|
249
95
|
}
|
|
250
|
-
// Login-wall hint (trawl_node#1975 `failureKind==='auth'`) — trust node's
|
|
251
|
-
// verdict, never re-detect the login URL client-side. Hypothesis-framed
|
|
252
|
-
// ("likely fix"), not a promise: measured cases (Reddit/X/Instagram) had
|
|
253
|
-
// valid cookies and stayed walled. Gated on `scrapId` alone (via
|
|
254
|
-
// `authHintWillRender` above, shared with the generic error line so the
|
|
255
|
-
// two can't drift apart), NOT the `scrapId ?? run._id` fallback used
|
|
256
|
-
// elsewhere here — `run._id` is a history id, and the session-set command
|
|
257
|
-
// on one silently targets the wrong document.
|
|
258
|
-
//
|
|
259
|
-
// trawl_cli#182 — excludes `conflictingSignals`: when a wall-vendor match
|
|
260
|
-
// also fired, the combined hedged line above already covers the cookie
|
|
261
|
-
// suggestion. Rendering this plain, unhedged block too would stack a
|
|
262
|
-
// second verdict under the first (and directly contradict it, since this
|
|
263
|
-
// block asserts the run IS a login wall with no caveat) — exactly the
|
|
264
|
-
// two-verdicts-for-one-run defect this fix removes. Exactly one verdict
|
|
265
|
-
// block renders per run.
|
|
266
96
|
if (authHintWillRender && !conflictingSignals) {
|
|
267
97
|
lines.push('');
|
|
268
98
|
lines.push(chalk.yellow(' Login wall (hypothesis): ') + 'this looks like a login redirect — your own session cookies are the likely fix.');
|
|
269
99
|
lines.push(chalk.dim(` → trawl scraps account session set ${scrapId} -c cookies.json (or app Settings → Account)`));
|
|
270
100
|
}
|
|
271
|
-
// Regression detail. trawl_node#1950 dropped the redundant `regressionDetected`
|
|
272
|
-
// boolean — it was `true` on every row iff `statusDetail === 'regression'`,
|
|
273
|
-
// so read that directly (works against a pre- or post-#1950 backend alike,
|
|
274
|
-
// no CLI/backend sequencing gap: `statusDetail` isn't part of this rename).
|
|
275
101
|
if (run.statusDetail === 'regression') {
|
|
276
102
|
lines.push(chalk.dim(' Regression: ') + `length ${run.length ?? '?'} vs baseline ${run.baselineLength ?? '?'}`);
|
|
277
103
|
}
|
|
278
|
-
// Autofix summary (when fix exists)
|
|
279
104
|
if (fix) {
|
|
280
105
|
lines.push(` ${chalk.magenta('Autofix:')} ${fix.outcome ?? '?'}${fix.classification ? ` — ${fix.classification}` : ''}${fix.reason ? ` (${fix.reason})` : ''}`);
|
|
281
106
|
lines.push(chalk.dim(` → full diff/dry-run/knowledge: trawl scraps autofix ${scrapId ?? run._id} (or doctor --autofix)`));
|
|
282
107
|
}
|
|
283
|
-
// Timestamp
|
|
284
108
|
if (run.createdAt) {
|
|
285
109
|
lines.push(chalk.dim(' Run at: ') + formatDate(run.createdAt));
|
|
286
110
|
}
|
|
287
|
-
// Snapshot hint
|
|
288
111
|
if (run.errorSnapshot?.html || run.statusDetail === 'empty' || run.statusDetail === 'error') {
|
|
289
112
|
lines.push('');
|
|
290
113
|
lines.push(chalk.dim(` → trawl scraps snapshot ${scrapId ?? run._id} --error (download error-path HTML)`));
|
|
291
114
|
}
|
|
292
115
|
return lines.join('\n');
|
|
293
116
|
}
|
|
294
|
-
/**
|
|
295
|
-
* Pure formatter — returns a human-readable autofix detail block.
|
|
296
|
-
* Shows diff, dry-run results, knowledge consulted.
|
|
297
|
-
*/
|
|
298
117
|
export function formatAutofix(fix) {
|
|
299
118
|
const lines = [];
|
|
300
119
|
lines.push(chalk.magenta.bold('Autofix attempt'));
|
|
@@ -324,17 +143,12 @@ export function formatAutofix(fix) {
|
|
|
324
143
|
}
|
|
325
144
|
return lines.join('\n');
|
|
326
145
|
}
|
|
327
|
-
/**
|
|
328
|
-
* Fetch the latest run + its ai_fix_end activity for a scrap.
|
|
329
|
-
* Returns null if the scrap has no runs yet.
|
|
330
|
-
*/
|
|
331
146
|
export async function fetchRunAndFix(scrapId) {
|
|
332
147
|
const scrap = await api.get(`/api/scraps/${scrapId}`);
|
|
333
148
|
const hid = scrap.history?.[0]?._id;
|
|
334
149
|
if (!hid)
|
|
335
150
|
return null;
|
|
336
151
|
const run = await api.get(`/api/historys/${hid}`);
|
|
337
|
-
// api.get unwraps the { data: T } envelope — the response is the array directly
|
|
338
152
|
const acts = await api.get(`/api/scraps/${scrapId}/activities?history=${hid}&limit=5`);
|
|
339
153
|
const fixAct = Array.isArray(acts) ? acts.find((a) => a.type === 'ai_fix_end') : undefined;
|
|
340
154
|
const fix = fixAct ? { ...fixAct.metadata, createdAt: fixAct.createdAt } : null;
|