crawlwise 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +55 -0
- package/bin/crawlwise.mjs +154 -0
- package/package.json +18 -0
package/README.md
ADDED
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# crawlwise CLI
|
|
2
|
+
|
|
3
|
+
Audit a URL from a terminal, and ask whether AI crawlers can reach it.
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
npx crawlwise ai-crawlers https://example.com
|
|
7
|
+
npx crawlwise audit https://example.com
|
|
8
|
+
npx crawlwise audit https://example.com --baseline <audit-id> --fail-on critical
|
|
9
|
+
npx crawlwise tools
|
|
10
|
+
```
|
|
11
|
+
|
|
12
|
+
## What needs a key and what does not
|
|
13
|
+
|
|
14
|
+
`ai-crawlers` and `tools` are free and need no key: they call the same endpoints the free
|
|
15
|
+
web tools call. `audit` needs an API key because an audit spends a credit, and Pro or above
|
|
16
|
+
can create one at <https://crawlwise.site/account/keys>.
|
|
17
|
+
|
|
18
|
+
```bash
|
|
19
|
+
export CRAWLWISE_API_KEY=cw_live_...
|
|
20
|
+
```
|
|
21
|
+
|
|
22
|
+
## Exit codes
|
|
23
|
+
|
|
24
|
+
- `0` — done, and no regression at or above `--fail-on` (default `critical`)
|
|
25
|
+
- `1` — a regression at or above `--fail-on`
|
|
26
|
+
- `2` — usage error, or the API refused the request
|
|
27
|
+
|
|
28
|
+
That makes it usable as a merge gate:
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
npx crawlwise audit "$PREVIEW_URL" --baseline "$LAST_KNOWN_GOOD_AUDIT" --fail-on critical
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
A check that did not run on the baseline is reported as new and is never counted as a
|
|
35
|
+
regression — a methodology change is not a defect.
|
|
36
|
+
|
|
37
|
+
## Options
|
|
38
|
+
|
|
39
|
+
| Flag | Meaning |
|
|
40
|
+
|---|---|
|
|
41
|
+
| `--json` | machine-readable output |
|
|
42
|
+
| `--api-key <key>` | instead of `CRAWLWISE_API_KEY` |
|
|
43
|
+
| `--endpoint <origin>` | point at staging, default `https://crawlwise.site` |
|
|
44
|
+
| `--fail-on <level>` | `critical`, `warning`, `never` |
|
|
45
|
+
|
|
46
|
+
## Publishing
|
|
47
|
+
|
|
48
|
+
This package is not on npm yet. Until it is, run it from a checkout:
|
|
49
|
+
|
|
50
|
+
```bash
|
|
51
|
+
node cli/bin/crawlwise.mjs ai-crawlers https://example.com
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
Publishing it is `npm publish` from this directory with `private` removed from
|
|
55
|
+
`package.json` — a deliberate step, and npm now requires 2FA-capable publishing.
|
|
@@ -0,0 +1,154 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
|
|
3
|
+
const args = process.argv.slice(2);
|
|
4
|
+
const flag = (name, fallback = '') => {
|
|
5
|
+
const index = args.indexOf('--' + name);
|
|
6
|
+
if (index < 0) return fallback;
|
|
7
|
+
const next = args[index + 1];
|
|
8
|
+
return next && !next.startsWith('--') ? next : 'true';
|
|
9
|
+
};
|
|
10
|
+
const has = (name) => args.includes('--' + name);
|
|
11
|
+
const positional = args.filter((arg, index) => !arg.startsWith('--') && !(index > 0 && args[index - 1].startsWith('--') && !args[index - 1].includes('=')));
|
|
12
|
+
|
|
13
|
+
const endpoint = (flag('endpoint') || process.env.CRAWLWISE_ENDPOINT || 'https://crawlwise.site').replace(/\/+$/, '');
|
|
14
|
+
const apiKey = flag('api-key') || process.env.CRAWLWISE_API_KEY || '';
|
|
15
|
+
const asJson = has('json');
|
|
16
|
+
|
|
17
|
+
function usage() {
|
|
18
|
+
console.log(`crawlwise
|
|
19
|
+
|
|
20
|
+
npx crawlwise audit <url> audit one page
|
|
21
|
+
npx crawlwise audit <url> --baseline <id> audit and diff against an earlier audit
|
|
22
|
+
npx crawlwise ai-crawlers <url> can GPTBot, ClaudeBot and the rest reach it? free, no key
|
|
23
|
+
npx crawlwise tools the free toolkit, with what each one cannot do
|
|
24
|
+
|
|
25
|
+
Options
|
|
26
|
+
--json machine-readable output
|
|
27
|
+
--api-key <key> or set CRAWLWISE_API_KEY (cw_live_…, Pro plan or above)
|
|
28
|
+
--endpoint <origin> or set CRAWLWISE_ENDPOINT (default ${endpoint})
|
|
29
|
+
--fail-on <level> critical | warning | never (audit diffs only, default critical)
|
|
30
|
+
--help
|
|
31
|
+
|
|
32
|
+
Exit codes: 0 ok · 1 regression at or above --fail-on · 2 usage or request error`);
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
function fail(message, code = 2) {
|
|
36
|
+
console.error('crawlwise: ' + message);
|
|
37
|
+
process.exit(code);
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
async function api(path, init = {}) {
|
|
41
|
+
const response = await fetch(endpoint + path, {
|
|
42
|
+
...init,
|
|
43
|
+
headers: { 'Content-Type': 'application/json', ...(apiKey ? { Authorization: 'Bearer ' + apiKey } : {}), ...(init.headers ?? {}) },
|
|
44
|
+
});
|
|
45
|
+
const body = await response.json().catch(() => ({}));
|
|
46
|
+
if (!response.ok) {
|
|
47
|
+
const detail = body?.error?.message || body?.error || 'HTTP ' + response.status;
|
|
48
|
+
fail(detail, 2);
|
|
49
|
+
}
|
|
50
|
+
return body;
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
async function audit(url, baselineId) {
|
|
54
|
+
if (!apiKey) fail('an audit needs an API key. Set CRAWLWISE_API_KEY, or run: crawlwise ai-crawlers <url> (free, no key).');
|
|
55
|
+
const started = await api('/v1/audits', { method: 'POST', body: JSON.stringify({ url }) });
|
|
56
|
+
process.stderr.write('crawlwise: audit ' + started.id + ' started\n');
|
|
57
|
+
const deadline = Date.now() + 5 * 60_000;
|
|
58
|
+
let job;
|
|
59
|
+
while (Date.now() < deadline) {
|
|
60
|
+
await new Promise((resolve) => setTimeout(resolve, 2500));
|
|
61
|
+
job = await api('/v1/audits/' + encodeURIComponent(started.id));
|
|
62
|
+
if (job.status === 'complete' || job.status === 'failed' || job.status === 'abandoned') break;
|
|
63
|
+
}
|
|
64
|
+
if (!job || job.status !== 'complete') fail('the audit ended as ' + (job?.status ?? 'still running') + (job?.error ? ': ' + job.error : ''), 2);
|
|
65
|
+
|
|
66
|
+
if (asJson) {
|
|
67
|
+
console.log(JSON.stringify({ id: started.id, status: job.status, url: job.url, report: job.report }, null, 2));
|
|
68
|
+
return 0;
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
const report = job.report ?? {};
|
|
72
|
+
const scoring = report.scoring ?? {};
|
|
73
|
+
console.log('');
|
|
74
|
+
console.log((job.url || url) + ' · ' + (scoring.score === null || scoring.score === undefined ? 'not scored' : 'score ' + scoring.score + '/' + (scoring.applicable ?? '?') + ' applicable'));
|
|
75
|
+
const priority = report.priority ?? [];
|
|
76
|
+
const review = report.worth_considering ?? [];
|
|
77
|
+
const unmeasured = report.not_measured ?? [];
|
|
78
|
+
console.log(priority.length + ' to fix · ' + review.length + ' worth reviewing · ' + unmeasured.length + ' not measured');
|
|
79
|
+
for (const [index, action] of priority.slice(0, 10).entries()) {
|
|
80
|
+
console.log('');
|
|
81
|
+
console.log(' ' + (index + 1) + '. [' + action.priority + '] ' + action.title);
|
|
82
|
+
console.log(' ' + action.why);
|
|
83
|
+
}
|
|
84
|
+
console.log('');
|
|
85
|
+
console.log('Report: ' + endpoint + '/history');
|
|
86
|
+
|
|
87
|
+
if (!baselineId) return 0;
|
|
88
|
+
const before = await api('/v1/audits/' + encodeURIComponent(baselineId));
|
|
89
|
+
if (before.status !== 'complete') fail('the baseline audit ' + baselineId + ' is ' + before.status, 2);
|
|
90
|
+
const old = new Map((before.report?.checks ?? []).map((check) => [check.id, check]));
|
|
91
|
+
const regressions = (report.checks ?? []).filter((check) => check.status !== 'pass' && old.get(check.id)?.status === 'pass');
|
|
92
|
+
console.log('');
|
|
93
|
+
console.log(regressions.length ? regressions.length + ' regression(s) against ' + baselineId + ':' : 'No regressions against ' + baselineId + '.');
|
|
94
|
+
for (const check of regressions) console.log(' ' + check.id + ': pass → ' + check.status + ' ' + check.title);
|
|
95
|
+
|
|
96
|
+
const failOn = flag('fail-on', 'critical');
|
|
97
|
+
if (failOn === 'never') return 0;
|
|
98
|
+
const critical = regressions.filter((check) => ['index', 'ai-crawler-access'].includes(check.id));
|
|
99
|
+
if (failOn === 'critical' && critical.length) return 1;
|
|
100
|
+
if (failOn === 'warning' && regressions.length) return 1;
|
|
101
|
+
return 0;
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
async function aiCrawlers(url) {
|
|
105
|
+
const body = await api('/api/tools/ai-crawlers', { method: 'POST', body: JSON.stringify({ url }) });
|
|
106
|
+
if (asJson) {
|
|
107
|
+
console.log(JSON.stringify(body, null, 2));
|
|
108
|
+
return 0;
|
|
109
|
+
}
|
|
110
|
+
console.log('');
|
|
111
|
+
console.log(body.finalUrl || url);
|
|
112
|
+
for (const row of body.rows ?? []) {
|
|
113
|
+
console.log(' ' + row.id.padEnd(20) + row.verdict.padEnd(17) + (row.deliberate ? '(deliberate opt-out)' : row.purpose));
|
|
114
|
+
}
|
|
115
|
+
const blocked = (body.rows ?? []).filter((row) => row.verdict === 'blocked' && !row.deliberate);
|
|
116
|
+
if (blocked.length) {
|
|
117
|
+
console.log('');
|
|
118
|
+
console.log(blocked.length + ' crawler(s) blocked without an intended opt-out: ' + blocked.map((row) => row.id).join(', '));
|
|
119
|
+
}
|
|
120
|
+
if (body.shareId) console.log('\nShareable: ' + endpoint + '/tools/ai-crawler-check/r/' + body.shareId);
|
|
121
|
+
return 0;
|
|
122
|
+
}
|
|
123
|
+
|
|
124
|
+
async function tools() {
|
|
125
|
+
const body = await api('/api/tools/catalogue').catch(() => null);
|
|
126
|
+
const list = body?.tools ?? [];
|
|
127
|
+
if (asJson) {
|
|
128
|
+
console.log(JSON.stringify(list, null, 2));
|
|
129
|
+
return 0;
|
|
130
|
+
}
|
|
131
|
+
console.log('');
|
|
132
|
+
console.log('Free tools, no account, no credit: ' + endpoint + '/tools');
|
|
133
|
+
for (const tool of list) console.log(' ' + tool.slug.padEnd(22) + tool.title);
|
|
134
|
+
return 0;
|
|
135
|
+
}
|
|
136
|
+
|
|
137
|
+
const [command, target] = positional;
|
|
138
|
+
if (!command || has('help') || command === 'help') {
|
|
139
|
+
usage();
|
|
140
|
+
process.exit(has('help') || command === 'help' ? 0 : 2);
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
if (command === 'audit') {
|
|
144
|
+
if (!target) fail('which URL? crawlwise audit https://example.com');
|
|
145
|
+
process.exit(await audit(target, flag('baseline')));
|
|
146
|
+
} else if (command === 'ai-crawlers') {
|
|
147
|
+
if (!target) fail('which URL? crawlwise ai-crawlers https://example.com');
|
|
148
|
+
process.exit(await aiCrawlers(target));
|
|
149
|
+
} else if (command === 'tools') {
|
|
150
|
+
process.exit(await tools());
|
|
151
|
+
} else {
|
|
152
|
+
usage();
|
|
153
|
+
process.exit(2);
|
|
154
|
+
}
|
package/package.json
ADDED
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "crawlwise",
|
|
3
|
+
"version": "1.0.0",
|
|
4
|
+
"description": "Audit a URL from the command line, without installing anything.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"bin": { "crawlwise": "bin/crawlwise.mjs" },
|
|
7
|
+
"engines": { "node": ">=22" },
|
|
8
|
+
"files": ["bin", "README.md"],
|
|
9
|
+
"license": "UNLICENSED",
|
|
10
|
+
"repository": {
|
|
11
|
+
"type": "git",
|
|
12
|
+
"url": "git+https://github.com/Adnanarodiya/crawlwise.git",
|
|
13
|
+
"directory": "cli"
|
|
14
|
+
},
|
|
15
|
+
"homepage": "https://crawlwise.site/agents",
|
|
16
|
+
"bugs": "https://github.com/Adnanarodiya/crawlwise/issues",
|
|
17
|
+
"keywords": ["seo", "audit", "ai-crawlers", "gptbot", "cli", "crawlwise"]
|
|
18
|
+
}
|