bwb-browser 4.0.0 → 4.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +43 -17
- package/README.md +88 -40
- package/lib/act.mjs +474 -215
- package/lib/browser.mjs +189 -32
- package/lib/config.mjs +180 -0
- package/lib/diagnose.mjs +13 -11
- package/lib/fetch.mjs +187 -75
- package/lib/fingerprint.mjs +73 -28
- package/lib/helpers.mjs +86 -39
- package/lib/session.mjs +34 -7
- package/lib/setup.mjs +94 -13
- package/lib/tabs.mjs +127 -15
- package/lib/urlpolicy.mjs +137 -0
- package/lib/vigil.mjs +44 -21
- package/package.json +15 -7
- package/server.mjs +433 -213
package/lib/session.mjs
CHANGED
|
@@ -7,20 +7,37 @@
|
|
|
7
7
|
* Sessions are stored as JSON files in ~/.bwb/sessions/
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
|
-
import { mkdirSync, writeFileSync, readFileSync, existsSync, readdirSync } from "fs";
|
|
10
|
+
import { mkdirSync, writeFileSync, readFileSync, existsSync, readdirSync, chmodSync } from "fs";
|
|
11
11
|
import { homedir } from "os";
|
|
12
12
|
import { join } from "path";
|
|
13
13
|
|
|
14
14
|
const SESSION_DIR = join(homedir(), ".bwb", "sessions");
|
|
15
15
|
|
|
16
|
+
// These files hold live login cookies for every domain the browser has visited.
|
|
17
|
+
// They were written with the default umask (0644 = world-readable) inside a
|
|
18
|
+
// 0755 directory. 600/700, and chmod pre-existing files on read too.
|
|
16
19
|
function ensureDir() {
|
|
17
|
-
mkdirSync(SESSION_DIR, { recursive: true });
|
|
20
|
+
mkdirSync(SESSION_DIR, { recursive: true, mode: 0o700 });
|
|
21
|
+
try { chmodSync(SESSION_DIR, 0o700); } catch {}
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
function writeSecure(path, data) {
|
|
25
|
+
writeFileSync(path, data, { mode: 0o600 });
|
|
26
|
+
chmodSync(path, 0o600);
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
/** True when a cookie belongs to one of the requested domains. */
|
|
30
|
+
function inDomains(cookie, domains) {
|
|
31
|
+
if (!domains || !domains.length) return true;
|
|
32
|
+
const host = String(cookie.domain || "").replace(/^\./, "").toLowerCase();
|
|
33
|
+
return domains.some((d) => host === d || host.endsWith("." + d));
|
|
18
34
|
}
|
|
19
35
|
|
|
20
36
|
/**
|
|
21
37
|
* Save all cookies from the current browser session to a named session file.
|
|
38
|
+
* @param {object} [opts] {domains} — restrict to specific domains
|
|
22
39
|
*/
|
|
23
|
-
export async function saveSession(name, protocol) {
|
|
40
|
+
export async function saveSession(name, protocol, { domains } = {}) {
|
|
24
41
|
if (!name || typeof name !== "string") {
|
|
25
42
|
throw new Error("Session name is required");
|
|
26
43
|
}
|
|
@@ -29,15 +46,25 @@ export async function saveSession(name, protocol) {
|
|
|
29
46
|
ensureDir();
|
|
30
47
|
|
|
31
48
|
const { cookies } = await protocol.Network.getAllCookies();
|
|
49
|
+
const wanted = domains?.length
|
|
50
|
+
? cookies.filter((c) => inDomains(c, domains.map((d) => String(d).toLowerCase())))
|
|
51
|
+
: cookies;
|
|
32
52
|
const path = join(SESSION_DIR, `${safeName}.json`);
|
|
33
|
-
|
|
53
|
+
writeSecure(path, JSON.stringify({
|
|
34
54
|
name: safeName,
|
|
35
|
-
cookieCount:
|
|
55
|
+
cookieCount: wanted.length,
|
|
56
|
+
domains: domains || null,
|
|
36
57
|
savedAt: Date.now(),
|
|
37
|
-
cookies,
|
|
58
|
+
cookies: wanted,
|
|
38
59
|
}, null, 2));
|
|
39
60
|
|
|
40
|
-
return {
|
|
61
|
+
return {
|
|
62
|
+
savedTo: path,
|
|
63
|
+
cookieCount: wanted.length,
|
|
64
|
+
name: safeName,
|
|
65
|
+
...(domains?.length ? { domains } : {}),
|
|
66
|
+
warning: "This file contains live login credentials (mode 600). Treat it like a password.",
|
|
67
|
+
};
|
|
41
68
|
}
|
|
42
69
|
|
|
43
70
|
/**
|
package/lib/setup.mjs
CHANGED
|
@@ -8,14 +8,23 @@
|
|
|
8
8
|
*/
|
|
9
9
|
|
|
10
10
|
import fs from 'fs';
|
|
11
|
+
import os from 'os';
|
|
11
12
|
import path from 'path';
|
|
12
13
|
import { fileURLToPath } from 'url';
|
|
13
|
-
import {
|
|
14
|
+
import { execFileSync } from 'child_process';
|
|
14
15
|
|
|
15
16
|
const __dirname = path.dirname(fileURLToPath(import.meta.url));
|
|
16
|
-
|
|
17
|
+
// os.homedir(), not process.env.HOME: on Windows HOME is usually unset, and the
|
|
18
|
+
// old '/root' fallback made setup look in a directory nobody has.
|
|
19
|
+
const HOME = os.homedir();
|
|
17
20
|
const SERVER_PATH = path.resolve(__dirname, '..', 'server.mjs');
|
|
18
21
|
|
|
22
|
+
// ─── Dry run by default ──────────────────────────────────────────────────────
|
|
23
|
+
// setup edits up to eight other tools' config files. It used to do that with no
|
|
24
|
+
// confirmation and no way to preview; --yes applies, --dry-run (default) prints.
|
|
25
|
+
const ASSUME_YES = process.argv.includes('--yes');
|
|
26
|
+
const DRY_RUN = process.argv.includes('--dry-run') || !ASSUME_YES;
|
|
27
|
+
|
|
19
28
|
// ─── Detect Chrome/Chromium ───────────────────────────────────────
|
|
20
29
|
function detectChrome() {
|
|
21
30
|
if (process.env.BWB_CHROME_PATH) {
|
|
@@ -34,18 +43,32 @@ function detectChrome() {
|
|
|
34
43
|
'C:\\Program Files (x86)\\Google\\Chrome\\Application\\chrome.exe',
|
|
35
44
|
];
|
|
36
45
|
|
|
46
|
+
// No shell: the candidate list is data. `which`/`where` is tried per bin.
|
|
47
|
+
const lookups = os.platform() === 'win32' ? ['where'] : ['which', 'command -v'];
|
|
37
48
|
for (const bin of candidates) {
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
43
|
-
|
|
49
|
+
if (path.isAbsolute(bin) && fs.existsSync(bin)) return { path: bin, source: 'filesystem' };
|
|
50
|
+
for (const cmd of lookups) {
|
|
51
|
+
try {
|
|
52
|
+
const args = cmd === 'command -v' ? ['-v', bin] : [bin];
|
|
53
|
+
const out = execFileSync(cmd, args, {
|
|
54
|
+
encoding: 'utf8', timeout: 3000, stdio: ['ignore', 'pipe', 'ignore']
|
|
55
|
+
}).split(/\r?\n/).map((l) => l.trim()).filter(Boolean)[0];
|
|
56
|
+
if (out) return { path: out, source: 'PATH' };
|
|
57
|
+
} catch { /* not found */ }
|
|
58
|
+
}
|
|
44
59
|
}
|
|
45
60
|
|
|
46
61
|
return null;
|
|
47
62
|
}
|
|
48
63
|
|
|
64
|
+
function whichExists(bin) {
|
|
65
|
+
const cmd = os.platform() === 'win32' ? 'where' : 'which';
|
|
66
|
+
try {
|
|
67
|
+
execFileSync(cmd, [bin], { stdio: 'ignore', timeout: 3000 });
|
|
68
|
+
return true;
|
|
69
|
+
} catch { return false; }
|
|
70
|
+
}
|
|
71
|
+
|
|
49
72
|
// ─── Detect AI CLI tools ──────────────────────────────────────────
|
|
50
73
|
const AGENT_CONFIGS = [
|
|
51
74
|
{
|
|
@@ -90,9 +113,26 @@ const AGENT_CONFIGS = [
|
|
|
90
113
|
}
|
|
91
114
|
},
|
|
92
115
|
{
|
|
116
|
+
// Claude Code reads MCP servers from ~/.claude.json (user scope) or a
|
|
117
|
+
// project .mcp.json — NOT from ~/.claude/settings.json, where the old
|
|
118
|
+
// setup wrote mcpServers and then reported "✅ configured" while nothing
|
|
119
|
+
// changed (anthropics/claude-code#4976, #26167). Use the CLI when it
|
|
120
|
+
// exists so the write lands where Claude Code actually looks.
|
|
93
121
|
name: 'Claude Code',
|
|
94
|
-
file: path.join(HOME, '.claude
|
|
122
|
+
file: path.join(HOME, '.claude.json'),
|
|
95
123
|
detect: (cfg) => !!(cfg.mcpServers?.bwb),
|
|
124
|
+
cli() {
|
|
125
|
+
const add = ['mcp', 'add', '--scope', 'user', 'bwb', '--', process.execPath, SERVER_PATH];
|
|
126
|
+
const print = `claude ${add.join(' ')}`;
|
|
127
|
+
if (!whichExists('claude')) return { manual: print };
|
|
128
|
+
if (DRY_RUN) return { manual: print };
|
|
129
|
+
try {
|
|
130
|
+
execFileSync('claude', add, { encoding: 'utf8', timeout: 30000, stdio: ['ignore', 'pipe', 'pipe'] });
|
|
131
|
+
return { ok: true, file: path.join(HOME, '.claude.json') };
|
|
132
|
+
} catch (err) {
|
|
133
|
+
return { manual: print, error: err.message };
|
|
134
|
+
}
|
|
135
|
+
},
|
|
96
136
|
add(cfg) {
|
|
97
137
|
if (!cfg.mcpServers) cfg.mcpServers = {};
|
|
98
138
|
cfg.mcpServers.bwb = {
|
|
@@ -162,7 +202,10 @@ const AGENT_CONFIGS = [
|
|
|
162
202
|
|
|
163
203
|
// ─── Core logic ────────────────────────────────────────────────────
|
|
164
204
|
function backup(file) {
|
|
165
|
-
|
|
205
|
+
// Timestamped: a single .bak was overwritten on every re-run, so the second
|
|
206
|
+
// run destroyed the only copy of the user's original config.
|
|
207
|
+
const stamp = new Date().toISOString().replace(/[:.]/g, '-').slice(0, 19);
|
|
208
|
+
const bak = `${file}.${stamp}.bak`;
|
|
166
209
|
try {
|
|
167
210
|
fs.copyFileSync(file, bak);
|
|
168
211
|
return bak;
|
|
@@ -179,7 +222,19 @@ function writeConfigSafely(file, data) {
|
|
|
179
222
|
}
|
|
180
223
|
|
|
181
224
|
function configureAgent(agent) {
|
|
182
|
-
const { name, file, detect, add } = agent;
|
|
225
|
+
const { name, file, detect, add, cli } = agent;
|
|
226
|
+
|
|
227
|
+
// Some agents must be configured through their own CLI, not by editing JSON.
|
|
228
|
+
if (cli) {
|
|
229
|
+
const res = cli();
|
|
230
|
+
if (res.ok) return { name, status: 'configured', file: res.file, via: 'claude mcp add' };
|
|
231
|
+
if (res.manual) {
|
|
232
|
+
return fs.existsSync(file) && detect(JSON.parse(fs.readFileSync(file, 'utf8')))
|
|
233
|
+
? { name, status: 'already configured', file }
|
|
234
|
+
: { name, status: 'manual', command: res.manual, error: res.error };
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
|
|
183
238
|
const fileExists = fs.existsSync(file);
|
|
184
239
|
|
|
185
240
|
if (!fileExists) return { name, status: 'skipped', reason: 'not installed' };
|
|
@@ -194,8 +249,11 @@ function configureAgent(agent) {
|
|
|
194
249
|
return { name, status: 'already configured', file };
|
|
195
250
|
}
|
|
196
251
|
|
|
197
|
-
const bak = backup(file);
|
|
198
252
|
config = add(config);
|
|
253
|
+
if (DRY_RUN) {
|
|
254
|
+
return { name, status: 'would configure', file };
|
|
255
|
+
}
|
|
256
|
+
const bak = backup(file);
|
|
199
257
|
if (!writeConfigSafely(file, config)) {
|
|
200
258
|
return { name, status: 'error', reason: 'write failed' };
|
|
201
259
|
}
|
|
@@ -263,12 +321,21 @@ export function runSetup() {
|
|
|
263
321
|
const results = AGENT_CONFIGS.map(configureAgent);
|
|
264
322
|
|
|
265
323
|
const configured = results.filter(r => r.status === 'configured');
|
|
324
|
+
const would = results.filter(r => r.status === 'would configure');
|
|
266
325
|
const alreadyDone = results.filter(r => r.status === 'already configured');
|
|
267
326
|
const skipped = results.filter(r => r.status === 'skipped');
|
|
268
327
|
const errors = results.filter(r => r.status === 'error');
|
|
269
328
|
|
|
270
329
|
for (const r of results) {
|
|
271
330
|
switch (r.status) {
|
|
331
|
+
case 'would configure':
|
|
332
|
+
console.log(` 📝 ${r.name}: would write ${r.file}`);
|
|
333
|
+
break;
|
|
334
|
+
case 'manual':
|
|
335
|
+
console.log(` 📎 ${r.name}: run this yourself — bwb cannot verify it for you:`);
|
|
336
|
+
console.log(` ${r.command}`);
|
|
337
|
+
if (r.error) console.log(` (attempt failed: ${String(r.error).slice(0, 120)})`);
|
|
338
|
+
break;
|
|
272
339
|
case 'configured':
|
|
273
340
|
console.log(` ✅ ${r.name}: configured`);
|
|
274
341
|
if (r.backup) console.log(` backup: ${r.backup}`);
|
|
@@ -289,9 +356,23 @@ export function runSetup() {
|
|
|
289
356
|
console.log(` ✅ Agents configured: ${configured.length}`);
|
|
290
357
|
console.log(` ✅ Already set up: ${alreadyDone.length}`);
|
|
291
358
|
console.log(` ⏭️ Not installed: ${skipped.length}`);
|
|
359
|
+
if (would.length) console.log(` 📝 Would configure: ${would.length} (re-run with --yes to apply)`);
|
|
292
360
|
if (errors.length) console.log(` ❌ Errors: ${errors.length}`);
|
|
293
361
|
console.log();
|
|
294
362
|
|
|
363
|
+
if (DRY_RUN && (would.length || configured.length)) {
|
|
364
|
+
console.log(' 🔎 DRY RUN — nothing was written.');
|
|
365
|
+
console.log(' Apply with: bwb --setup --yes\n');
|
|
366
|
+
}
|
|
367
|
+
if (DRY_RUN && !process.stdout.isTTY) {
|
|
368
|
+
// Not a terminal => almost certainly a provisioning script or CI step.
|
|
369
|
+
// A silent no-op here is the one 4.x -> 4.1.0 break that does not fail
|
|
370
|
+
// loudly, so say it on stderr where scripts actually surface errors.
|
|
371
|
+
console.error(
|
|
372
|
+
'bwb: --setup ran in dry-run mode and wrote NOTHING.\n' +
|
|
373
|
+
' If you are scripting this, pass --yes: bwb --setup --yes\n'
|
|
374
|
+
);
|
|
375
|
+
}
|
|
295
376
|
if (configured.length > 0) {
|
|
296
377
|
console.log(' 🔄 RESTART REQUIRED: Close and reopen your AI agent');
|
|
297
378
|
console.log(' for the new MCP tools to take effect.\n');
|
|
@@ -304,7 +385,7 @@ export function runSetup() {
|
|
|
304
385
|
console.log(' Linux: apt install chromium-browser\n');
|
|
305
386
|
}
|
|
306
387
|
|
|
307
|
-
console.log(' 🚀 Ready to go! Try: browser_status\n');
|
|
388
|
+
if (!DRY_RUN) console.log(' 🚀 Ready to go! Try: browser_status\n');
|
|
308
389
|
printSurvivalGuide();
|
|
309
390
|
return { configured: configured.length, alreadyDone: alreadyDone.length, errors: errors.length };
|
|
310
391
|
}
|
package/lib/tabs.mjs
CHANGED
|
@@ -7,19 +7,38 @@
|
|
|
7
7
|
*/
|
|
8
8
|
|
|
9
9
|
import CDP from "chrome-remote-interface";
|
|
10
|
-
import { ensureBrowser, protocol, actualCdpPort, cfg } from "./browser.mjs";
|
|
11
|
-
import {
|
|
10
|
+
import { ensureBrowser, protocol, actualCdpPort, cfg, attached, browserExited } from "./browser.mjs";
|
|
11
|
+
import { assertNavigable } from "./urlpolicy.mjs";
|
|
12
|
+
import { writeFileSync, readFileSync, existsSync, mkdirSync, chmodSync } from "fs";
|
|
12
13
|
import { join } from "path";
|
|
13
14
|
|
|
14
15
|
// ─── Tab Journal (working-set survival across LMK kills / mayfly teardown) ───
|
|
15
16
|
// Tiny JSON append on every mutation. Journal is the truth the next fresh
|
|
16
17
|
// browser restores from — cookies live in user-data-dir, tab URLs live here.
|
|
18
|
+
//
|
|
19
|
+
// The journal stores ORIGIN + PATH only. Tab URLs routinely carry OAuth
|
|
20
|
+
// callbacks, magic links and reset tokens in the query string, and on desktop
|
|
21
|
+
// nothing re-navigates them anyway (see restoreJournal) — so writing and
|
|
22
|
+
// replaying the full URL was leaking credentials onto disk for no benefit.
|
|
23
|
+
// BWB_JOURNAL=full restores the old behaviour for anyone who needs it.
|
|
17
24
|
|
|
18
25
|
function journalPath() {
|
|
19
26
|
if (!cfg.userDataDir) return null;
|
|
20
27
|
return join(cfg.userDataDir, "bwb-tabs.json");
|
|
21
28
|
}
|
|
22
29
|
|
|
30
|
+
/** Strip query + fragment unless the user explicitly asked for full URLs. */
|
|
31
|
+
export function journalUrl(url) {
|
|
32
|
+
if (!url) return url;
|
|
33
|
+
if (cfg.journalFull) return url;
|
|
34
|
+
try {
|
|
35
|
+
const u = new URL(url);
|
|
36
|
+
return u.origin === "null" ? u.href : `${u.origin}${u.pathname}`;
|
|
37
|
+
} catch {
|
|
38
|
+
return url.split(/[?#]/)[0];
|
|
39
|
+
}
|
|
40
|
+
}
|
|
41
|
+
|
|
23
42
|
function saveJournal() {
|
|
24
43
|
try {
|
|
25
44
|
const path = journalPath();
|
|
@@ -27,10 +46,17 @@ function saveJournal() {
|
|
|
27
46
|
const entries = [];
|
|
28
47
|
for (const [id, tab] of tabs) {
|
|
29
48
|
if (tab.url && tab.url !== "about:blank") {
|
|
30
|
-
entries.push({
|
|
49
|
+
entries.push({
|
|
50
|
+
url: journalUrl(tab.url),
|
|
51
|
+
...(cfg.journalFull ? { fullUrl: tab.url } : {}),
|
|
52
|
+
title: tab.title || "",
|
|
53
|
+
active: id === activeTabId,
|
|
54
|
+
});
|
|
31
55
|
}
|
|
32
56
|
}
|
|
33
|
-
|
|
57
|
+
mkdirSync(path.replace(/[^/\\]+$/, ""), { recursive: true, mode: 0o700 });
|
|
58
|
+
writeFileSync(path, JSON.stringify(entries), { mode: 0o600 });
|
|
59
|
+
chmodSync(path, 0o600);
|
|
34
60
|
} catch {}
|
|
35
61
|
}
|
|
36
62
|
|
|
@@ -39,7 +65,11 @@ function loadJournal() {
|
|
|
39
65
|
const path = journalPath();
|
|
40
66
|
if (!path || !existsSync(path)) return [];
|
|
41
67
|
const entries = JSON.parse(readFileSync(path, "utf8"));
|
|
42
|
-
|
|
68
|
+
if (!Array.isArray(entries)) return [];
|
|
69
|
+
// fullUrl only exists when BWB_JOURNAL=full was on when it was written.
|
|
70
|
+
return entries
|
|
71
|
+
.filter((e) => e && (e.url || e.fullUrl))
|
|
72
|
+
.map((e) => ({ ...e, url: e.fullUrl || e.url }));
|
|
43
73
|
} catch {
|
|
44
74
|
return [];
|
|
45
75
|
}
|
|
@@ -65,7 +95,12 @@ async function ensureDefaultTab() {
|
|
|
65
95
|
|
|
66
96
|
// Browser must be started first
|
|
67
97
|
await ensureBrowser();
|
|
68
|
-
if (!protocol)
|
|
98
|
+
if (!protocol) {
|
|
99
|
+
throw new Error(
|
|
100
|
+
`Browser not started (exited=${browserExited} attached=${attached} port=${actualCdpPort}). ` +
|
|
101
|
+
`If another bwb process is using ${cfg.userDataDir}, it kills this browser — use a different --user-data-dir or --port.`
|
|
102
|
+
);
|
|
103
|
+
}
|
|
69
104
|
|
|
70
105
|
const port = actualCdpPort || cfg.port;
|
|
71
106
|
const targets = await CDP.List({ port });
|
|
@@ -82,6 +117,11 @@ async function ensureDefaultTab() {
|
|
|
82
117
|
* Returns the CDP protocol for the active tab.
|
|
83
118
|
* If no tabs are managed, ensures browser is started and returns default protocol.
|
|
84
119
|
* All tool handlers should use this instead of ensureBrowser() directly.
|
|
120
|
+
*
|
|
121
|
+
* This is also where the static rung is paid off: if browser_goto answered
|
|
122
|
+
* from a plain HTTP fetch, the browser has never seen the URL. The first tool
|
|
123
|
+
* that needs Chromium materializes it here — the browser starts, navigates to
|
|
124
|
+
* the same URL, and every later tool works on that page.
|
|
85
125
|
*/
|
|
86
126
|
export async function getActiveProtocol() {
|
|
87
127
|
if (tabs.size === 0) {
|
|
@@ -98,26 +138,90 @@ export async function getActiveProtocol() {
|
|
|
98
138
|
return protocol;
|
|
99
139
|
}
|
|
100
140
|
|
|
141
|
+
/**
|
|
142
|
+
* A page served by the static rung that Chromium has not loaded yet.
|
|
143
|
+
* Set by browser_goto; consumed (navigated to) by the first browser tool.
|
|
144
|
+
*/
|
|
145
|
+
let pendingStatic = null;
|
|
146
|
+
|
|
147
|
+
export function setPendingStatic(url, title = "") {
|
|
148
|
+
pendingStatic = url ? { url, title } : null;
|
|
149
|
+
}
|
|
150
|
+
|
|
151
|
+
export function getPendingStatic() {
|
|
152
|
+
return pendingStatic;
|
|
153
|
+
}
|
|
154
|
+
|
|
155
|
+
export function clearPendingStatic() {
|
|
156
|
+
pendingStatic = null;
|
|
157
|
+
}
|
|
158
|
+
|
|
159
|
+
/** True when a browser-backed tool must run but Chromium shows nothing. */
|
|
160
|
+
export function needsMaterialize() {
|
|
161
|
+
if (!pendingStatic) return false;
|
|
162
|
+
const tab = activeTabId ? tabs.get(activeTabId) : null;
|
|
163
|
+
const current = tab?.url || "";
|
|
164
|
+
return !current || current === "about:blank" || current !== pendingStatic.url;
|
|
165
|
+
}
|
|
166
|
+
|
|
167
|
+
/**
|
|
168
|
+
* Bring Chromium to the page the static rung served, if the tab is not already
|
|
169
|
+
* there. Called from the single choke point (server.mjs) before any tool that
|
|
170
|
+
* needs a live page.
|
|
171
|
+
*/
|
|
172
|
+
export async function materializeStatic() {
|
|
173
|
+
if (!needsMaterialize()) return null;
|
|
174
|
+
const target = pendingStatic;
|
|
175
|
+
const cdp = await getActiveProtocol();
|
|
176
|
+
try {
|
|
177
|
+
assertNavigable(target.url);
|
|
178
|
+
await cdp.Page.enable?.();
|
|
179
|
+
// Wait for the real load events rather than a fixed sleep: a fixed sleep
|
|
180
|
+
// is how the first browser_text after a static goto comes back empty.
|
|
181
|
+
const { gotoUrl } = await import("./helpers.mjs");
|
|
182
|
+
const landed = await gotoUrl(cdp.Page, cdp.Runtime, target.url, cfg.navTimeout || 30000);
|
|
183
|
+
pendingStatic = null;
|
|
184
|
+
syncActiveTab(landed.title, landed.url);
|
|
185
|
+
return landed;
|
|
186
|
+
} catch (err) {
|
|
187
|
+
pendingStatic = null;
|
|
188
|
+
throw err;
|
|
189
|
+
}
|
|
190
|
+
}
|
|
191
|
+
|
|
101
192
|
/**
|
|
102
193
|
* Hibernate a tab: URL stays in the journal + visible in listTabs, but the
|
|
103
194
|
* renderer is closed and memory freed. Woken transparently on switchTab.
|
|
104
195
|
* Never hibernates the active tab — caller must switch away first.
|
|
196
|
+
*
|
|
197
|
+
* The default tab's `protocol` IS the global browser connection, so closing
|
|
198
|
+
* that target used to leave the global connected to a dead target: the next
|
|
199
|
+
* ensureBrowser() saw a live `protocol` and handed out a connection to a tab
|
|
200
|
+
* that no longer existed.
|
|
105
201
|
*/
|
|
106
202
|
export async function hibernateTab(targetId) {
|
|
107
203
|
const id = targetId || [...tabs.keys()].find((k) => k !== activeTabId && !tabs.get(k)?.hibernated);
|
|
108
204
|
if (!id || !tabs.has(id)) return null;
|
|
109
205
|
const tab = tabs.get(id);
|
|
110
206
|
if (tab.hibernated) return { id, hibernated: true };
|
|
207
|
+
// Guest mode: those are the user's real tabs. Never close one to save RAM.
|
|
208
|
+
if (attached) return { id, skipped: true, reason: "attach mode: never closes real tabs" };
|
|
111
209
|
const port = actualCdpPort || cfg.port;
|
|
112
210
|
try { await CDP.Close({ id, port }); } catch {}
|
|
113
|
-
|
|
114
|
-
try { await tab.protocol.close(); } catch {}
|
|
115
|
-
}
|
|
211
|
+
closeTabConnection(tab);
|
|
116
212
|
tabs.set(id, { protocol: null, title: tab.title, url: tab.url, hibernated: true });
|
|
117
213
|
saveJournal();
|
|
118
214
|
return { id, hibernated: true, url: tab.url };
|
|
119
215
|
}
|
|
120
216
|
|
|
217
|
+
/** Close a tab's CDP connection, including the shared global one. */
|
|
218
|
+
function closeTabConnection(tab) {
|
|
219
|
+
if (!tab?.protocol) return;
|
|
220
|
+
try {
|
|
221
|
+
if (tab.protocol === protocol) protocol.close().catch(() => {});
|
|
222
|
+
} catch {}
|
|
223
|
+
}
|
|
224
|
+
|
|
121
225
|
/** Wake a hibernated tab: fresh target, journaled URL re-navigated. */
|
|
122
226
|
async function wakeTab(targetId) {
|
|
123
227
|
const tab = tabs.get(targetId);
|
|
@@ -136,6 +240,12 @@ async function wakeTab(targetId) {
|
|
|
136
240
|
* Restore the journaled working set after a fresh spawn. First entry
|
|
137
241
|
* navigates now; the rest become about:blank placeholders woken on switch.
|
|
138
242
|
* Capped at cfg.tabMax so restore never re-spikes memory at startup.
|
|
243
|
+
*
|
|
244
|
+
* Auto-restore RE-NAVIGATES. On a lean profile that is the point (resurrect
|
|
245
|
+
* the working set after an LMK kill). On desktop it is not: a stale journal
|
|
246
|
+
* from yesterday would re-fire every URL the moment bwb starts. So on desktop
|
|
247
|
+
* the entries are surfaced as placeholders and nothing is fetched until the
|
|
248
|
+
* agent asks for it via browser_listTabs + browser_switchTab.
|
|
139
249
|
*/
|
|
140
250
|
export async function restoreJournal() {
|
|
141
251
|
const entries = loadJournal();
|
|
@@ -144,9 +254,10 @@ export async function restoreJournal() {
|
|
|
144
254
|
const cap = cfg.tabMax && cfg.tabMax > 0 ? cfg.tabMax : entries.length;
|
|
145
255
|
const wanted = entries.slice(0, Math.max(cap, 1));
|
|
146
256
|
const port = actualCdpPort || cfg.port;
|
|
257
|
+
const auto = Boolean(cfg.lean) || Boolean(cfg.journalFull);
|
|
147
258
|
let first = true;
|
|
148
259
|
for (const entry of wanted) {
|
|
149
|
-
if (first) {
|
|
260
|
+
if (first && auto) {
|
|
150
261
|
first = false;
|
|
151
262
|
try {
|
|
152
263
|
await protocol.Page.navigate({ url: entry.url });
|
|
@@ -154,6 +265,7 @@ export async function restoreJournal() {
|
|
|
154
265
|
} catch {}
|
|
155
266
|
continue;
|
|
156
267
|
}
|
|
268
|
+
first = false;
|
|
157
269
|
try {
|
|
158
270
|
const info = await CDP.New({ port, url: "about:blank" });
|
|
159
271
|
const newProtocol = await CDP({ target: info.id, port });
|
|
@@ -161,7 +273,7 @@ export async function restoreJournal() {
|
|
|
161
273
|
} catch {}
|
|
162
274
|
}
|
|
163
275
|
saveJournal();
|
|
164
|
-
return
|
|
276
|
+
return auto;
|
|
165
277
|
}
|
|
166
278
|
|
|
167
279
|
// ─── Create Tab ──────────────────────────────────────────────────────────────
|
|
@@ -171,11 +283,13 @@ export async function restoreJournal() {
|
|
|
171
283
|
* Switches to the new tab automatically.
|
|
172
284
|
*/
|
|
173
285
|
export async function createTab(url) {
|
|
286
|
+
if (url) assertNavigable(url, { allowDomains: cfg.allowDomains || null });
|
|
174
287
|
await ensureDefaultTab();
|
|
175
288
|
|
|
176
289
|
// Live-tab cap: hibernate the oldest non-active tab instead of growing
|
|
177
290
|
// renderers until Android LMK notices. Journal keeps everything restorable.
|
|
178
|
-
|
|
291
|
+
// Never in attach mode — those tabs belong to the human.
|
|
292
|
+
if (!attached && cfg.tabMax && cfg.tabMax > 0) {
|
|
179
293
|
while (liveCount() >= cfg.tabMax) {
|
|
180
294
|
const victim = [...tabs.keys()].find((k) => k !== activeTabId && !tabs.get(k)?.hibernated);
|
|
181
295
|
if (!victim) break;
|
|
@@ -215,9 +329,7 @@ export async function closeTab(targetId) {
|
|
|
215
329
|
await CDP.Close({ id, port });
|
|
216
330
|
|
|
217
331
|
const tab = tabs.get(id);
|
|
218
|
-
|
|
219
|
-
try { await tab.protocol.close(); } catch {}
|
|
220
|
-
}
|
|
332
|
+
closeTabConnection(tab);
|
|
221
333
|
tabs.delete(id);
|
|
222
334
|
|
|
223
335
|
// Switch to another tab
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* bwb-browser — URL navigation policy
|
|
3
|
+
*
|
|
4
|
+
* One chokepoint for every URL that reaches the network or the browser. The
|
|
5
|
+
* agent reads untrusted pages, and a page can talk the agent into fetching
|
|
6
|
+
* `file:///…` or the cloud metadata address. Scheme and address checks live
|
|
7
|
+
* here so goto, newTab, act-navigation, download and static fetch all agree.
|
|
8
|
+
*/
|
|
9
|
+
|
|
10
|
+
const BLOCKED_SCHEMES = new Set([
|
|
11
|
+
"file:", "chrome:", "chrome-extension:", "chrome-search:", "chrome-untrusted:",
|
|
12
|
+
"devtools:", "javascript:", "view-source:", "data:", "blob:", "ftp:", "ws:", "wss:",
|
|
13
|
+
]);
|
|
14
|
+
|
|
15
|
+
const NAVIGABLE_SCHEMES = new Set(["http:", "https:", "about:"]);
|
|
16
|
+
|
|
17
|
+
export class UrlPolicyError extends Error {
|
|
18
|
+
constructor(message, url) {
|
|
19
|
+
super(message);
|
|
20
|
+
this.name = "UrlPolicyError";
|
|
21
|
+
this.url = url;
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
/**
|
|
26
|
+
* Parse and check a URL for navigation. Scheme allowlist only — this is what
|
|
27
|
+
* every navigation path (goto, newTab, act, history) goes through.
|
|
28
|
+
* @returns {URL}
|
|
29
|
+
*/
|
|
30
|
+
export function assertNavigable(raw, { allowDomains = null } = {}) {
|
|
31
|
+
if (typeof raw !== "string" || !raw.trim()) {
|
|
32
|
+
throw new UrlPolicyError("URL is required", raw);
|
|
33
|
+
}
|
|
34
|
+
let u;
|
|
35
|
+
try {
|
|
36
|
+
u = new URL(raw.trim());
|
|
37
|
+
} catch {
|
|
38
|
+
throw new UrlPolicyError(`Not a valid URL: ${raw}`, raw);
|
|
39
|
+
}
|
|
40
|
+
if (BLOCKED_SCHEMES.has(u.protocol)) {
|
|
41
|
+
throw new UrlPolicyError(`Blocked URL scheme: ${u.protocol}`, raw);
|
|
42
|
+
}
|
|
43
|
+
if (!NAVIGABLE_SCHEMES.has(u.protocol)) {
|
|
44
|
+
throw new UrlPolicyError(`Unsupported URL scheme: ${u.protocol} (http, https, about)`, raw);
|
|
45
|
+
}
|
|
46
|
+
if (allowDomains && allowDomains.length) {
|
|
47
|
+
assertDomainAllowed(u.hostname, allowDomains);
|
|
48
|
+
}
|
|
49
|
+
return u;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
/** http(s) only — for anything that will be fetched by us or a subprocess. */
|
|
53
|
+
export function assertHttpUrl(raw, opts = {}) {
|
|
54
|
+
const u = assertNavigable(raw, opts);
|
|
55
|
+
if (u.protocol !== "http:" && u.protocol !== "https:") {
|
|
56
|
+
throw new UrlPolicyError(`Only http(s) URLs are allowed, got ${u.protocol}`, raw);
|
|
57
|
+
}
|
|
58
|
+
return u;
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* Check a hostname against an optional allowlist. `example.com` also covers
|
|
63
|
+
* `www.example.com`; `*.example.com` covers subdomains only.
|
|
64
|
+
*/
|
|
65
|
+
export function assertDomainAllowed(hostname, allowDomains) {
|
|
66
|
+
if (!allowDomains || !allowDomains.length) return true;
|
|
67
|
+
const host = String(hostname || "").toLowerCase();
|
|
68
|
+
const ok = allowDomains.some((d) => host === d || host.endsWith("." + d));
|
|
69
|
+
if (!ok) {
|
|
70
|
+
throw new UrlPolicyError(`Domain not allowed by --allow-domains: ${host}`, host);
|
|
71
|
+
}
|
|
72
|
+
return true;
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
const IPV4_RE = /^(\d{1,3})\.(\d{1,3})\.(\d{1,3})\.(\d{1,3})$/;
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* Is this hostname a loopback / private / link-local address or a name that
|
|
79
|
+
* resolves to one? DNS names other than localhost are not resolved here (the
|
|
80
|
+
* static fetch resolves them itself), but literal IPs and known-local names
|
|
81
|
+
* are caught without a lookup.
|
|
82
|
+
*/
|
|
83
|
+
export function isPrivateHost(hostname) {
|
|
84
|
+
const host = String(hostname || "").toLowerCase().replace(/^\[|\]$/g, "");
|
|
85
|
+
if (!host) return true;
|
|
86
|
+
if (host === "localhost" || host.endsWith(".localhost") || host.endsWith(".local")) return true;
|
|
87
|
+
if (host === "metadata.google.internal" || host.endsWith(".internal")) return true;
|
|
88
|
+
|
|
89
|
+
const v4 = IPV4_RE.exec(host);
|
|
90
|
+
if (v4) {
|
|
91
|
+
const [a, b] = [Number(v4[1]), Number(v4[2])];
|
|
92
|
+
if (a === 127) return true; // 127/8 loopback
|
|
93
|
+
if (a === 10) return true; // 10/8
|
|
94
|
+
if (a === 172 && b >= 16 && b <= 31) return true; // 172.16/12
|
|
95
|
+
if (a === 192 && b === 168) return true; // 192.168/16
|
|
96
|
+
if (a === 169 && b === 254) return true; // 169.254/16 link-local + metadata
|
|
97
|
+
if (a === 0) return true; // 0.0.0.0/8
|
|
98
|
+
if (a === 100 && b >= 64 && b <= 127) return true; // 100.64/10 CGNAT
|
|
99
|
+
if (a >= 224) return true; // multicast / reserved
|
|
100
|
+
return false;
|
|
101
|
+
}
|
|
102
|
+
|
|
103
|
+
if (host.includes(":")) {
|
|
104
|
+
// IPv6 literal
|
|
105
|
+
if (host === "::1" || host === "::") return true;
|
|
106
|
+
if (/^f[cd]/.test(host)) return true; // fc00::/7 unique-local
|
|
107
|
+
if (host.startsWith("fe80")) return true; // link-local
|
|
108
|
+
}
|
|
109
|
+
return false;
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
/**
|
|
113
|
+
* Full outbound-fetch policy: http(s) only, and (unless BWB_ALLOW_PRIVATE=1)
|
|
114
|
+
* no loopback / private / link-local targets. Used by the static fetch rung so
|
|
115
|
+
* a prompt-injected page cannot make the agent read the host's LAN or the
|
|
116
|
+
* cloud metadata service.
|
|
117
|
+
*
|
|
118
|
+
* `allowPrivate` may be `true` (allow everything private — local development)
|
|
119
|
+
* or an array of hostnames (only those). Tests and dev servers use the array
|
|
120
|
+
* form so a redirect to 169.254.169.254 is still refused.
|
|
121
|
+
*/
|
|
122
|
+
export async function assertOutbound(raw, { allowPrivate = false, allowDomains = null } = {}) {
|
|
123
|
+
const u = assertHttpUrl(raw, { allowDomains });
|
|
124
|
+
if (allowPrivate === true) return u;
|
|
125
|
+
const allowed = Array.isArray(allowPrivate)
|
|
126
|
+
? allowPrivate.map((h) => String(h).toLowerCase())
|
|
127
|
+
: [];
|
|
128
|
+
if (allowed.includes(u.hostname.toLowerCase())) return u;
|
|
129
|
+
if (isPrivateHost(u.hostname)) {
|
|
130
|
+
throw new UrlPolicyError(
|
|
131
|
+
`Refusing to fetch a private/loopback address (${u.hostname}). ` +
|
|
132
|
+
`Set BWB_ALLOW_PRIVATE=1 to allow local development targets.`,
|
|
133
|
+
raw
|
|
134
|
+
);
|
|
135
|
+
}
|
|
136
|
+
return u;
|
|
137
|
+
}
|