crawlwise 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md ADDED
@@ -0,0 +1,55 @@
1
+ # crawlwise CLI
2
+
3
+ Audit a URL from a terminal, and ask whether AI crawlers can reach it.
4
+
5
+ ```bash
6
+ npx crawlwise ai-crawlers https://example.com
7
+ npx crawlwise audit https://example.com
8
+ npx crawlwise audit https://example.com --baseline <audit-id> --fail-on critical
9
+ npx crawlwise tools
10
+ ```
11
+
12
+ ## What needs a key and what does not
13
+
14
+ `ai-crawlers` and `tools` are free and need no key: they call the same endpoints the free
15
+ web tools call. `audit` needs an API key because an audit spends a credit, and Pro or above
16
+ can create one at <https://crawlwise.site/account/keys>.
17
+
18
+ ```bash
19
+ export CRAWLWISE_API_KEY=cw_live_...
20
+ ```
21
+
22
+ ## Exit codes
23
+
24
+ - `0` — done, and no regression at or above `--fail-on` (default `critical`)
25
+ - `1` — a regression at or above `--fail-on`
26
+ - `2` — usage error, or the API refused the request
27
+
28
+ That makes it usable as a merge gate:
29
+
30
+ ```bash
31
+ npx crawlwise audit "$PREVIEW_URL" --baseline "$LAST_KNOWN_GOOD_AUDIT" --fail-on critical
32
+ ```
33
+
34
+ A check that did not run on the baseline is reported as new and is never counted as a
35
+ regression — a methodology change is not a defect.
36
+
37
+ ## Options
38
+
39
+ | Flag | Meaning |
40
+ |---|---|
41
+ | `--json` | machine-readable output |
42
+ | `--api-key <key>` | instead of `CRAWLWISE_API_KEY` |
43
+ | `--endpoint <origin>` | point at staging, default `https://crawlwise.site` |
44
+ | `--fail-on <level>` | `critical`, `warning`, `never` |
45
+
46
+ ## Publishing
47
+
48
+ This package is not on npm yet. Until it is, run it from a checkout:
49
+
50
+ ```bash
51
+ node cli/bin/crawlwise.mjs ai-crawlers https://example.com
52
+ ```
53
+
54
+ Publishing it is `npm publish` from this directory with `private` removed from
55
+ `package.json` — a deliberate step, and npm now requires 2FA-capable publishing.
@@ -0,0 +1,154 @@
1
+ #!/usr/bin/env node
2
+
3
+ const args = process.argv.slice(2);
4
+ const flag = (name, fallback = '') => {
5
+ const index = args.indexOf('--' + name);
6
+ if (index < 0) return fallback;
7
+ const next = args[index + 1];
8
+ return next && !next.startsWith('--') ? next : 'true';
9
+ };
10
+ const has = (name) => args.includes('--' + name);
11
+ const positional = args.filter((arg, index) => !arg.startsWith('--') && !(index > 0 && args[index - 1].startsWith('--') && !args[index - 1].includes('=')));
12
+
13
+ const endpoint = (flag('endpoint') || process.env.CRAWLWISE_ENDPOINT || 'https://crawlwise.site').replace(/\/+$/, '');
14
+ const apiKey = flag('api-key') || process.env.CRAWLWISE_API_KEY || '';
15
+ const asJson = has('json');
16
+
17
+ function usage() {
18
+ console.log(`crawlwise
19
+
20
+ npx crawlwise audit <url> audit one page
21
+ npx crawlwise audit <url> --baseline <id> audit and diff against an earlier audit
22
+ npx crawlwise ai-crawlers <url> can GPTBot, ClaudeBot and the rest reach it? free, no key
23
+ npx crawlwise tools the free toolkit, with what each one cannot do
24
+
25
+ Options
26
+ --json machine-readable output
27
+ --api-key <key> or set CRAWLWISE_API_KEY (cw_live_…, Pro plan or above)
28
+ --endpoint <origin> or set CRAWLWISE_ENDPOINT (default ${endpoint})
29
+ --fail-on <level> critical | warning | never (audit diffs only, default critical)
30
+ --help
31
+
32
+ Exit codes: 0 ok · 1 regression at or above --fail-on · 2 usage or request error`);
33
+ }
34
+
35
+ function fail(message, code = 2) {
36
+ console.error('crawlwise: ' + message);
37
+ process.exit(code);
38
+ }
39
+
40
+ async function api(path, init = {}) {
41
+ const response = await fetch(endpoint + path, {
42
+ ...init,
43
+ headers: { 'Content-Type': 'application/json', ...(apiKey ? { Authorization: 'Bearer ' + apiKey } : {}), ...(init.headers ?? {}) },
44
+ });
45
+ const body = await response.json().catch(() => ({}));
46
+ if (!response.ok) {
47
+ const detail = body?.error?.message || body?.error || 'HTTP ' + response.status;
48
+ fail(detail, 2);
49
+ }
50
+ return body;
51
+ }
52
+
53
+ async function audit(url, baselineId) {
54
+ if (!apiKey) fail('an audit needs an API key. Set CRAWLWISE_API_KEY, or run: crawlwise ai-crawlers <url> (free, no key).');
55
+ const started = await api('/v1/audits', { method: 'POST', body: JSON.stringify({ url }) });
56
+ process.stderr.write('crawlwise: audit ' + started.id + ' started\n');
57
+ const deadline = Date.now() + 5 * 60_000;
58
+ let job;
59
+ while (Date.now() < deadline) {
60
+ await new Promise((resolve) => setTimeout(resolve, 2500));
61
+ job = await api('/v1/audits/' + encodeURIComponent(started.id));
62
+ if (job.status === 'complete' || job.status === 'failed' || job.status === 'abandoned') break;
63
+ }
64
+ if (!job || job.status !== 'complete') fail('the audit ended as ' + (job?.status ?? 'still running') + (job?.error ? ': ' + job.error : ''), 2);
65
+
66
+ if (asJson) {
67
+ console.log(JSON.stringify({ id: started.id, status: job.status, url: job.url, report: job.report }, null, 2));
68
+ return 0;
69
+ }
70
+
71
+ const report = job.report ?? {};
72
+ const scoring = report.scoring ?? {};
73
+ console.log('');
74
+ console.log((job.url || url) + ' · ' + (scoring.score === null || scoring.score === undefined ? 'not scored' : 'score ' + scoring.score + '/' + (scoring.applicable ?? '?') + ' applicable'));
75
+ const priority = report.priority ?? [];
76
+ const review = report.worth_considering ?? [];
77
+ const unmeasured = report.not_measured ?? [];
78
+ console.log(priority.length + ' to fix · ' + review.length + ' worth reviewing · ' + unmeasured.length + ' not measured');
79
+ for (const [index, action] of priority.slice(0, 10).entries()) {
80
+ console.log('');
81
+ console.log(' ' + (index + 1) + '. [' + action.priority + '] ' + action.title);
82
+ console.log(' ' + action.why);
83
+ }
84
+ console.log('');
85
+ console.log('Report: ' + endpoint + '/history');
86
+
87
+ if (!baselineId) return 0;
88
+ const before = await api('/v1/audits/' + encodeURIComponent(baselineId));
89
+ if (before.status !== 'complete') fail('the baseline audit ' + baselineId + ' is ' + before.status, 2);
90
+ const old = new Map((before.report?.checks ?? []).map((check) => [check.id, check]));
91
+ const regressions = (report.checks ?? []).filter((check) => check.status !== 'pass' && old.get(check.id)?.status === 'pass');
92
+ console.log('');
93
+ console.log(regressions.length ? regressions.length + ' regression(s) against ' + baselineId + ':' : 'No regressions against ' + baselineId + '.');
94
+ for (const check of regressions) console.log(' ' + check.id + ': pass → ' + check.status + ' ' + check.title);
95
+
96
+ const failOn = flag('fail-on', 'critical');
97
+ if (failOn === 'never') return 0;
98
+ const critical = regressions.filter((check) => ['index', 'ai-crawler-access'].includes(check.id));
99
+ if (failOn === 'critical' && critical.length) return 1;
100
+ if (failOn === 'warning' && regressions.length) return 1;
101
+ return 0;
102
+ }
103
+
104
+ async function aiCrawlers(url) {
105
+ const body = await api('/api/tools/ai-crawlers', { method: 'POST', body: JSON.stringify({ url }) });
106
+ if (asJson) {
107
+ console.log(JSON.stringify(body, null, 2));
108
+ return 0;
109
+ }
110
+ console.log('');
111
+ console.log(body.finalUrl || url);
112
+ for (const row of body.rows ?? []) {
113
+ console.log(' ' + row.id.padEnd(20) + row.verdict.padEnd(17) + (row.deliberate ? '(deliberate opt-out)' : row.purpose));
114
+ }
115
+ const blocked = (body.rows ?? []).filter((row) => row.verdict === 'blocked' && !row.deliberate);
116
+ if (blocked.length) {
117
+ console.log('');
118
+ console.log(blocked.length + ' crawler(s) blocked without an intended opt-out: ' + blocked.map((row) => row.id).join(', '));
119
+ }
120
+ if (body.shareId) console.log('\nShareable: ' + endpoint + '/tools/ai-crawler-check/r/' + body.shareId);
121
+ return 0;
122
+ }
123
+
124
+ async function tools() {
125
+ const body = await api('/api/tools/catalogue').catch(() => null);
126
+ const list = body?.tools ?? [];
127
+ if (asJson) {
128
+ console.log(JSON.stringify(list, null, 2));
129
+ return 0;
130
+ }
131
+ console.log('');
132
+ console.log('Free tools, no account, no credit: ' + endpoint + '/tools');
133
+ for (const tool of list) console.log(' ' + tool.slug.padEnd(22) + tool.title);
134
+ return 0;
135
+ }
136
+
137
+ const [command, target] = positional;
138
+ if (!command || has('help') || command === 'help') {
139
+ usage();
140
+ process.exit(has('help') || command === 'help' ? 0 : 2);
141
+ }
142
+
143
+ if (command === 'audit') {
144
+ if (!target) fail('which URL? crawlwise audit https://example.com');
145
+ process.exit(await audit(target, flag('baseline')));
146
+ } else if (command === 'ai-crawlers') {
147
+ if (!target) fail('which URL? crawlwise ai-crawlers https://example.com');
148
+ process.exit(await aiCrawlers(target));
149
+ } else if (command === 'tools') {
150
+ process.exit(await tools());
151
+ } else {
152
+ usage();
153
+ process.exit(2);
154
+ }
package/package.json ADDED
@@ -0,0 +1,18 @@
1
+ {
2
+ "name": "crawlwise",
3
+ "version": "1.0.0",
4
+ "description": "Audit a URL from the command line, without installing anything.",
5
+ "type": "module",
6
+ "bin": { "crawlwise": "bin/crawlwise.mjs" },
7
+ "engines": { "node": ">=22" },
8
+ "files": ["bin", "README.md"],
9
+ "license": "UNLICENSED",
10
+ "repository": {
11
+ "type": "git",
12
+ "url": "git+https://github.com/Adnanarodiya/crawlwise.git",
13
+ "directory": "cli"
14
+ },
15
+ "homepage": "https://crawlwise.site/agents",
16
+ "bugs": "https://github.com/Adnanarodiya/crawlwise/issues",
17
+ "keywords": ["seo", "audit", "ai-crawlers", "gptbot", "cli", "crawlwise"]
18
+ }