strom-research 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +373 -0
- package/README.md +142 -0
- package/assets/lang/cs.json +302 -0
- package/assets/lang/de.json +302 -0
- package/assets/method/core.md +43 -0
- package/assets/method/enrich.md +11 -0
- package/assets/method/intake.md +30 -0
- package/assets/method/link.md +28 -0
- package/assets/method/locate.md +28 -0
- package/assets/method/narrate.md +13 -0
- package/assets/method/reading.md +62 -0
- package/assets/method/recording.md +59 -0
- package/assets/method/request.md +10 -0
- package/assets/method/verify.md +17 -0
- package/assets/plugins/README.md +23 -0
- package/assets/plugins/connectors/DISCOVERY.md +159 -0
- package/assets/plugins/connectors/README.md +376 -0
- package/assets/plugins/connectors/sdk.ts +168 -0
- package/assets/plugins/connectors/template.ts +38 -0
- package/assets/plugins/gitignore +4 -0
- package/dist/agents/files.js +313 -0
- package/dist/agents/global.js +257 -0
- package/dist/agents/launch.js +36 -0
- package/dist/agents/profiles.js +95 -0
- package/dist/brief/brief.js +345 -0
- package/dist/cli/commit.js +44 -0
- package/dist/cli/context.js +311 -0
- package/dist/cli/execute.js +154 -0
- package/dist/cli/fixes.js +78 -0
- package/dist/cli/format.js +53 -0
- package/dist/cli/help.js +59 -0
- package/dist/cli/main.js +152 -0
- package/dist/cli/menu.js +212 -0
- package/dist/cli/registry.js +96 -0
- package/dist/cli/ui.js +266 -0
- package/dist/cli/wizard.js +142 -0
- package/dist/cli.js +14 -0
- package/dist/commands/analysis.js +622 -0
- package/dist/commands/batch.js +181 -0
- package/dist/commands/checks.js +153 -0
- package/dist/commands/connectors.js +1377 -0
- package/dist/commands/guide.js +160 -0
- package/dist/commands/index.js +19 -0
- package/dist/commands/intake.js +234 -0
- package/dist/commands/media.js +406 -0
- package/dist/commands/meta.js +195 -0
- package/dist/commands/output.js +117 -0
- package/dist/commands/people.js +664 -0
- package/dist/commands/read.js +199 -0
- package/dist/commands/research.js +139 -0
- package/dist/commands/session.js +605 -0
- package/dist/commands/setup.js +465 -0
- package/dist/commands/sources.js +634 -0
- package/dist/commands/start.js +383 -0
- package/dist/commands/story.js +75 -0
- package/dist/commands/tasks.js +436 -0
- package/dist/commands/trees.js +128 -0
- package/dist/core/actions.js +852 -0
- package/dist/core/age.js +95 -0
- package/dist/core/apps.js +74 -0
- package/dist/core/assets.js +34 -0
- package/dist/core/awake.js +33 -0
- package/dist/core/browser.js +281 -0
- package/dist/core/calibration.js +48 -0
- package/dist/core/check.js +112 -0
- package/dist/core/chromium.js +88 -0
- package/dist/core/config.js +348 -0
- package/dist/core/connector.js +811 -0
- package/dist/core/deps.js +73 -0
- package/dist/core/dialog.js +61 -0
- package/dist/core/errors.js +89 -0
- package/dist/core/evidence.js +58 -0
- package/dist/core/frontier.js +219 -0
- package/dist/core/gdate.js +77 -0
- package/dist/core/git.js +300 -0
- package/dist/core/guard.js +124 -0
- package/dist/core/http2.js +76 -0
- package/dist/core/import.js +541 -0
- package/dist/core/install.js +28 -0
- package/dist/core/integrity.js +219 -0
- package/dist/core/json.js +87 -0
- package/dist/core/lang.js +70 -0
- package/dist/core/live.js +244 -0
- package/dist/core/lock.js +112 -0
- package/dist/core/logins.js +67 -0
- package/dist/core/media.js +223 -0
- package/dist/core/model.js +101 -0
- package/dist/core/net.js +366 -0
- package/dist/core/open.js +29 -0
- package/dist/core/paths.js +84 -0
- package/dist/core/people.js +283 -0
- package/dist/core/phrases.js +85 -0
- package/dist/core/queue.js +113 -0
- package/dist/core/reader.js +76 -0
- package/dist/core/records.js +105 -0
- package/dist/core/roles.js +30 -0
- package/dist/core/schema.js +261 -0
- package/dist/core/seal.js +77 -0
- package/dist/core/self.js +40 -0
- package/dist/core/session.js +155 -0
- package/dist/core/shortcut.js +90 -0
- package/dist/core/stories.js +61 -0
- package/dist/core/stromapp.js +138 -0
- package/dist/core/text.js +104 -0
- package/dist/core/tree.js +507 -0
- package/dist/core/uninstall.js +128 -0
- package/dist/core/update.js +193 -0
- package/dist/core/validate.js +260 -0
- package/dist/core/views.js +164 -0
- package/dist/core/which.js +51 -0
- package/dist/core/workers.js +42 -0
- package/dist/gedcom/export.js +454 -0
- package/dist/gedcom/labels.js +103 -0
- package/dist/gedcom/lines.js +91 -0
- package/dist/gedcom/parse.js +53 -0
- package/dist/gedcom/validate.js +183 -0
- package/dist/image/image.js +223 -0
- package/dist/image/index.js +114 -0
- package/dist/image/jpeg-decode.js +552 -0
- package/dist/image/jpeg-encode.js +254 -0
- package/dist/image/png.js +241 -0
- package/dist/runners/antigravity.js +70 -0
- package/dist/runners/claude.js +179 -0
- package/dist/runners/codex.js +45 -0
- package/dist/runners/index.js +13 -0
- package/dist/runners/jsonl.js +86 -0
- package/dist/runners/opencode.js +50 -0
- package/dist/runners/runner.js +63 -0
- package/dist/runners/script.js +58 -0
- package/package.json +44 -0
|
@@ -0,0 +1,1377 @@
|
|
|
1
|
+
// connector list · show · new · add · remove · test — fetch — allow connector · allow host · consents · login
|
|
2
|
+
//
|
|
3
|
+
// Archive downloads through plugins (connectors) in the plugins folder,
|
|
4
|
+
// <shared>/plugins/connectors/<name>/: copied in by the user, or built there
|
|
5
|
+
// with their agent. strom paces every request, keeps to the hosts a connector
|
|
6
|
+
// names, writes the files itself, checks the images and registers them with
|
|
7
|
+
// their provenance. When the user asks to be asked (connectors.consent on), a
|
|
8
|
+
// connector runs only with their consent, given on a terminal after a plain
|
|
9
|
+
// warning — never by an agent; code that goes round strom needs it either way.
|
|
10
|
+
import fs from "node:fs";
|
|
11
|
+
import os from "node:os";
|
|
12
|
+
import path from "node:path";
|
|
13
|
+
import { ui } from "../cli/ui.js";
|
|
14
|
+
import { register } from "../cli/registry.js";
|
|
15
|
+
import { lines, moreLine, paginate, runs, shellArg, table } from "../cli/format.js";
|
|
16
|
+
import { StromError, UsageError } from "../core/errors.js";
|
|
17
|
+
import { requireRecord } from "../core/records.js";
|
|
18
|
+
import { readable } from "../core/text.js";
|
|
19
|
+
import { findImage, regionText, sameRegion } from "../core/media.js";
|
|
20
|
+
import { partRegion } from "../core/views.js";
|
|
21
|
+
import { loadLogins, loginOf, loginsFile, removeLogin, saveLogin, VISIBLE_FIELDS } from "../core/logins.js";
|
|
22
|
+
import { readAsset } from "../core/assets.js";
|
|
23
|
+
import { runGit } from "../core/git.js";
|
|
24
|
+
import { imageSizeOfFile } from "../image/index.js";
|
|
25
|
+
import { Tree } from "../core/tree.js";
|
|
26
|
+
import { syncAgentFiles } from "../agents/files.js";
|
|
27
|
+
import { clearBlock, CookieJar, DEFAULT_PACE, hostAllowed, hostState, knownHosts, NetError, paceOf, politeRequest, refusedBy, reserveSlots } from "../core/net.js";
|
|
28
|
+
import { botCheck, fileBase, findDownload, loadPage, loadPlans, pageKey, pageScript, parseResult, PLAN_TTL_MS, planScript, readSavedPage, removePlan, savePage, savePlan, SCRIPT_MS, } from "../core/browser.js";
|
|
29
|
+
import { bareHost, changedSinceConsent, codeWarning, consentRequired, connectorHash, connectorsDir, directNetwork, ensurePluginsDir, findConnector, hostWarning, imageProblem, INTERFACE, listConnectors, loadConsents, MANIFEST, missingConsents, NAME_RE, readManifest, routeOf, routesOf, ROUTES, runConnector, saveConsents, scanConnectors, suggestName, } from "../core/connector.js";
|
|
30
|
+
import { registerImages } from "./media.js";
|
|
31
|
+
function shared(ctx) {
|
|
32
|
+
const s = ctx.settings.shared();
|
|
33
|
+
if (!s)
|
|
34
|
+
throw new StromError("Strom is not set up yet", { hint: "strom setup" });
|
|
35
|
+
return s.value;
|
|
36
|
+
}
|
|
37
|
+
const netDir = (ctx) => path.join(shared(ctx), "net");
|
|
38
|
+
const when = (ms) => new Date(ms).toISOString().slice(0, 16).replace("T", " ");
|
|
39
|
+
function hostLine(ctx, host) {
|
|
40
|
+
const s = hostState(netDir(ctx), host);
|
|
41
|
+
const hour = s.recent.filter((t) => t > Date.now() - 3600_000).length;
|
|
42
|
+
return [
|
|
43
|
+
host,
|
|
44
|
+
s.blockedUntil && s.blockedUntil > Date.now() ? `⛔ refused us — left alone until ${when(s.blockedUntil)} (${s.reason ?? ""})` : "",
|
|
45
|
+
hour ? `${hour} request(s) in the last hour` : "",
|
|
46
|
+
(s.slowdown ?? 1) > 1 ? `slowed ×${(s.slowdown ?? 1).toFixed(1)}` : "",
|
|
47
|
+
]
|
|
48
|
+
.filter(Boolean)
|
|
49
|
+
.join(" · ");
|
|
50
|
+
}
|
|
51
|
+
/** Ask the user, on a terminal, to allow automated access to a host. */
|
|
52
|
+
async function askHost(ctx, host, c, byWindow = false) {
|
|
53
|
+
if (!byWindow) {
|
|
54
|
+
ctx.io.stdout("\n" + hostWarning(host, c) + "\n");
|
|
55
|
+
if (!(await ctx.confirm(`Allow automated access to ${host}?`, false)))
|
|
56
|
+
return false;
|
|
57
|
+
}
|
|
58
|
+
const all = loadConsents(ctx.env);
|
|
59
|
+
all.hosts[bareHost(host)] = { at: new Date().toISOString(), ...(c ? { connector: c.name } : {}), ...(c?.manifest.policy.terms ? { terms: c.manifest.policy.terms } : {}) };
|
|
60
|
+
saveConsents(ctx.env, all);
|
|
61
|
+
return true;
|
|
62
|
+
}
|
|
63
|
+
/** Everything a window asks the person about a connector at once: what it does, where it goes, any warning. */
|
|
64
|
+
function consentWindow(ctx, c) {
|
|
65
|
+
const lang = ctx.uiLang();
|
|
66
|
+
const miss = missingConsents(ctx.env, c);
|
|
67
|
+
return [
|
|
68
|
+
ui(lang, "ui.consent.connector", { name: c.manifest.title }),
|
|
69
|
+
ui(lang, "ui.consent.connector.hosts", { hosts: c.manifest.hosts.map(bareHost).join(", ") }),
|
|
70
|
+
...(miss.code === "changed" ? [ui(lang, "ui.consent.connector.changed")] : []),
|
|
71
|
+
...(directNetwork(c).length ? [ui(lang, "ui.consent.connector.direct")] : []),
|
|
72
|
+
...(c.manifest.policy.terms ? [ui(lang, "ui.consent.connector.terms", { terms: c.manifest.policy.terms })] : []),
|
|
73
|
+
].join("\n\n");
|
|
74
|
+
}
|
|
75
|
+
/** The user's yes to a connector's folder (and its code as it is now). */
|
|
76
|
+
function grant(ctx, c, source) {
|
|
77
|
+
const all = loadConsents(ctx.env);
|
|
78
|
+
all.connectors[c.name] = { dir: c.dir, hash: connectorHash(c.dir), hosts: c.manifest.hosts, at: new Date().toISOString(), source };
|
|
79
|
+
saveConsents(ctx.env, all);
|
|
80
|
+
}
|
|
81
|
+
/**
|
|
82
|
+
* What is left to allow: on the terminal the connector (after its warning), then each host.
|
|
83
|
+
* By a window the person said yes to all of it at once (the window showed it all: consentWindow).
|
|
84
|
+
*/
|
|
85
|
+
async function askConnector(ctx, c, source, byWindow = false) {
|
|
86
|
+
const miss = missingConsents(ctx.env, c);
|
|
87
|
+
if (miss.code) {
|
|
88
|
+
if (!byWindow) {
|
|
89
|
+
ctx.io.stdout("\n" + codeWarning(c, source, directNetwork(c)) + "\n");
|
|
90
|
+
if (!(await ctx.confirm(`${miss.code === "changed" ? "Its code changed — allow" : "Allow"} connector ${c.name}?`, false)))
|
|
91
|
+
return { allowed: false, hosts: [], missing: c.manifest.hosts.map(bareHost) };
|
|
92
|
+
}
|
|
93
|
+
grant(ctx, c, source);
|
|
94
|
+
}
|
|
95
|
+
const hosts = [];
|
|
96
|
+
const ask = consentRequired(ctx.env);
|
|
97
|
+
for (const h of c.manifest.hosts.map(bareHost))
|
|
98
|
+
if (!ask || loadConsents(ctx.env).hosts[h] || (await askHost(ctx, h, c, byWindow)))
|
|
99
|
+
hosts.push(h);
|
|
100
|
+
return { allowed: true, hosts, missing: c.manifest.hosts.map(bareHost).filter((h) => !hosts.includes(h)) };
|
|
101
|
+
}
|
|
102
|
+
/**
|
|
103
|
+
* Before a connector runs: everything allowed? The user, on a terminal, is asked
|
|
104
|
+
* right here; anyone else (an agent, a script) gets exit 4 and the command for the user.
|
|
105
|
+
*/
|
|
106
|
+
async function ensureAllowed(ctx, c) {
|
|
107
|
+
const miss = missingConsents(ctx.env, c);
|
|
108
|
+
if (!miss.code && !miss.hosts.length)
|
|
109
|
+
return true;
|
|
110
|
+
const changed = miss.code === "changed";
|
|
111
|
+
const how = ctx.requireHuman(consentRequired(ctx.env)
|
|
112
|
+
? `${changed ? "The code of" : "Run"} connector ${c.name} (${c.manifest.title})${changed ? " changed since you allowed it — allow it again" : ""}${miss.hosts.length ? `, with automated access to ${miss.hosts.join(", ")}` : ""}?`
|
|
113
|
+
: `Connector ${c.name} (${c.manifest.title}) reaches the network itself, past strom's limiter${changed ? ", and its code changed since you allowed it" : ""} — run it?`, `strom allow connector ${c.name}`, `connector:${c.name}`, consentWindow(ctx, c));
|
|
114
|
+
const r = await askConnector(ctx, c, ctx.display(c.dir), how === "window");
|
|
115
|
+
return r.allowed && !r.missing.length;
|
|
116
|
+
}
|
|
117
|
+
function consentState(ctx, c) {
|
|
118
|
+
const miss = missingConsents(ctx.env, c);
|
|
119
|
+
if (!miss.code && !miss.hosts.length)
|
|
120
|
+
return !consentRequired(ctx.env) ? "ready" : changedSinceConsent(ctx.env, c) ? "allowed (changed since)" : "allowed";
|
|
121
|
+
return [miss.code === "new" ? "needs your consent" : miss.code === "changed" ? "code changed — needs your consent again" : "", miss.hosts.length ? `hosts not allowed: ${miss.hosts.join(", ")}` : ""].filter(Boolean).join("; ");
|
|
122
|
+
}
|
|
123
|
+
/** " · 4317×3233 px" of an image file, when strom can read its size. */
|
|
124
|
+
function pixels(file) {
|
|
125
|
+
const s = imageSizeOfFile(file);
|
|
126
|
+
return s ? ` ${s.width}×${s.height} px` : "";
|
|
127
|
+
}
|
|
128
|
+
function describeRun(r, what) {
|
|
129
|
+
return lines(`${what}: ${r.requests} request(s)${r.pages ? ` + ${r.pages} page(s) your browser got` : ""}${r.stopped ? ` · stopped: ${r.stopped}` : ""}`, ...r.logs.slice(-10).map((l) => ` · ${l}`));
|
|
130
|
+
}
|
|
131
|
+
/** Up to this many books, each comes with the command to register it; more, one line each. */
|
|
132
|
+
const BOOKS_IN_FULL = 3;
|
|
133
|
+
const BOOKS_SHOWN = 30;
|
|
134
|
+
function bookLines(r, connector, listOne) {
|
|
135
|
+
const line = (b) => `${b.title}${b.callNumber ? ` (${b.callNumber})` : ""}${b.years ? ` · ${b.years}` : ""}${b.kinds?.length ? ` · ${b.kinds.join(",")}` : ""}${b.images ? ` · ${b.images} images` : ""}`;
|
|
136
|
+
if (r.books.length > BOOKS_IN_FULL)
|
|
137
|
+
return [
|
|
138
|
+
...r.books.slice(0, BOOKS_SHOWN).map((b) => ` ${b.id ? `${b.id} ` : ""}${line(b)}`),
|
|
139
|
+
r.books.length > BOOKS_SHOWN ? ` … ${r.books.length - BOOKS_SHOWN} more — narrow it with --years, or read them all with --json` : undefined,
|
|
140
|
+
` one of them, with the command to register it: ${listOne}`,
|
|
141
|
+
].filter((l) => l !== undefined);
|
|
142
|
+
return r.books.map((b) => [
|
|
143
|
+
` ${line(b)}`,
|
|
144
|
+
b.url ? ` ${b.url}` : undefined,
|
|
145
|
+
` strom recordset add ${shellArg(b.title)}${b.callNumber ? ` --call-number ${shellArg(b.callNumber)}` : ""}${b.kinds?.length ? ` --kinds ${b.kinds.join(",")}` : ""}${b.places?.length ? ` --places ${shellArg(b.places.join(","))}` : ""}${b.years ? ` --years ${b.years}` : ""}${b.url ? ` --url ${shellArg(b.url)}` : ""} --access online-free`,
|
|
146
|
+
b.id ? ` then: strom fetch ${connector} ${shellArg(b.id)} --images <from-to> --recordset B…` : undefined,
|
|
147
|
+
]
|
|
148
|
+
.filter(Boolean)
|
|
149
|
+
.join("\n"));
|
|
150
|
+
}
|
|
151
|
+
function parseImages(v, max = 1000) {
|
|
152
|
+
if (v === undefined)
|
|
153
|
+
return undefined;
|
|
154
|
+
const m = /^(\d+)(?:-(\d+))?$/.exec(String(v).trim());
|
|
155
|
+
if (!m)
|
|
156
|
+
throw new UsageError(`--images must be n or from-to, not "${v}"`);
|
|
157
|
+
const a = Number(m[1]);
|
|
158
|
+
const z = Number(m[2] ?? m[1]);
|
|
159
|
+
if (a < 1)
|
|
160
|
+
throw new UsageError("--images: images are counted from 1");
|
|
161
|
+
if (z < a)
|
|
162
|
+
throw new UsageError("--images: from before to");
|
|
163
|
+
if (z - a >= max)
|
|
164
|
+
throw new UsageError(`--images: at most ${max} at a time`, { hint: "the hourly cap of an archive: the rest in the next hour" });
|
|
165
|
+
return Array.from({ length: z - a + 1 }, (_, i) => a + i);
|
|
166
|
+
}
|
|
167
|
+
/** An archive the tree says must not be automated (forbidden, or browser only). */
|
|
168
|
+
function forbiddenBy(ctx, c) {
|
|
169
|
+
const root = ctx.locateTree();
|
|
170
|
+
if (!root)
|
|
171
|
+
return undefined;
|
|
172
|
+
return ctx
|
|
173
|
+
.tree()
|
|
174
|
+
.list("repository")
|
|
175
|
+
.find((r) => {
|
|
176
|
+
if ((r.automation !== "forbidden" && r.automation !== "manual") || !r.url)
|
|
177
|
+
return false;
|
|
178
|
+
try {
|
|
179
|
+
return hostAllowed(new URL(r.url).hostname, c.manifest.hosts);
|
|
180
|
+
}
|
|
181
|
+
catch {
|
|
182
|
+
return false;
|
|
183
|
+
}
|
|
184
|
+
});
|
|
185
|
+
}
|
|
186
|
+
/** How long a fetch takes at least, at the connector's pace (one request per image). */
|
|
187
|
+
function estimate(c, images) {
|
|
188
|
+
const pace = paceOf(c.manifest.policy.pace);
|
|
189
|
+
const secs = Math.round(((images - 1) * pace.minIntervalMs) / 1000);
|
|
190
|
+
const took = secs < 90 ? `${secs} s` : `${Math.round(secs / 60)} min`;
|
|
191
|
+
return `${images} image(s) through ${c.name}: at least ${took} at its pace (one request per image, ≥${pace.minIntervalMs / 1000} s apart)`;
|
|
192
|
+
}
|
|
193
|
+
/** How a connector's images come here, and what else it can do. */
|
|
194
|
+
function routeText(ctx, c) {
|
|
195
|
+
const r = routeOf(ctx.env, c);
|
|
196
|
+
const other = routesOf(c).filter((x) => x !== r.via);
|
|
197
|
+
return [
|
|
198
|
+
r.via === "browser" ? "through your browser" : "directly, through strom",
|
|
199
|
+
r.chosen ? `(chosen ${r.chosen.at.slice(0, 10)} ${r.chosen.by})` : undefined,
|
|
200
|
+
other.length ? `· can also: ${other.join(", ")} (strom connector use ${c.name} --via ${other[0]})` : undefined,
|
|
201
|
+
]
|
|
202
|
+
.filter(Boolean)
|
|
203
|
+
.join(" ");
|
|
204
|
+
}
|
|
205
|
+
register({
|
|
206
|
+
path: ["connector", "list"],
|
|
207
|
+
summary: "Connectors: plugins that download from an archive portal — which there are, whether they can run",
|
|
208
|
+
group: "sources",
|
|
209
|
+
run(ctx) {
|
|
210
|
+
const s = ctx.settings.shared()?.value;
|
|
211
|
+
const { connectors, broken } = scanConnectors(s);
|
|
212
|
+
const folder = s ? `folder: ${ctx.display(connectorsDir(s))} — copy a connector's folder in to install it (the contract: README.md there)` : "not set up yet: strom setup";
|
|
213
|
+
const data = {
|
|
214
|
+
folder: s ? connectorsDir(s) : undefined,
|
|
215
|
+
connectors: connectors.map((c) => ({ name: c.name, dir: c.dir, title: c.manifest.title, can: c.manifest.can, hosts: c.manifest.hosts, policy: c.manifest.policy, consent: consentState(ctx, c), routes: routesOf(c), via: routeOf(ctx.env, c).via })),
|
|
216
|
+
broken,
|
|
217
|
+
};
|
|
218
|
+
if (!connectors.length && !broken.length)
|
|
219
|
+
return {
|
|
220
|
+
text: lines("no connectors yet", ` ${folder}`, " an archive the research needs has none: build its connector — strom connector new <name> --url <portal>"),
|
|
221
|
+
data,
|
|
222
|
+
};
|
|
223
|
+
return {
|
|
224
|
+
text: lines(connectors.length ? table(connectors.map((c) => [c.name, c.manifest.title, c.manifest.can.join(","), `automation ${c.manifest.policy.automation}`, consentState(ctx, c), routeOf(ctx.env, c).via === "browser" ? "via your browser" : ""])) : undefined, ...broken.map((b) => `${b.name} ⚠ cannot run: ${b.problem}`), folder),
|
|
225
|
+
data,
|
|
226
|
+
};
|
|
227
|
+
},
|
|
228
|
+
}, {
|
|
229
|
+
path: ["connector", "show"],
|
|
230
|
+
summary: "One connector: what it can do, what the portal allows, your consent, how its hosts are doing",
|
|
231
|
+
group: "sources",
|
|
232
|
+
args: [{ name: "connector", description: "its name (its folder's name)", required: true }],
|
|
233
|
+
examples: ["strom connector show example-archive"],
|
|
234
|
+
run(ctx, { args }) {
|
|
235
|
+
const c = findConnector(ctx.settings.shared()?.value, args[0]);
|
|
236
|
+
const m = c.manifest;
|
|
237
|
+
const direct = directNetwork(c);
|
|
238
|
+
const pace = paceOf(m.policy.pace);
|
|
239
|
+
return {
|
|
240
|
+
text: lines(`${c.name}${m.version ? ` ${m.version}` : ""} — ${m.title} (${ctx.display(c.dir)})`, `can ${m.can.join(", ")}`, `runs ${m.run.join(" ")}`, `automation ${m.policy.automation}${m.policy.terms ? ` · terms ${m.policy.terms}` : " · terms not recorded"}`, m.policy.termsSummary ? ` ${m.policy.termsSummary}` : undefined, m.policy.officialExport ? `official ${m.policy.officialExport}` : undefined, `pace ≥${pace.minIntervalMs / 1000} s apart, ≤${pace.perHour}/h`, m.can.includes("fetch") || m.can.includes("locate") ? `images ${routeText(ctx, c)}` : undefined, `consent ${consentState(ctx, c)}`, m.login ? `login ${loginState(ctx, c)} — ${m.login.about}${m.login.url ? ` (${m.login.url})` : ""}` : undefined, direct.length ? `⚠ reaches the network directly (strom cannot pace that):\n${direct.map((d) => ` ${d}`).join("\n")}` : "network only through strom (checked)", "hosts", ...m.hosts.map((h) => ` ${hostLine(ctx, bareHost(h))}`)),
|
|
241
|
+
data: { name: c.name, connector: m, dir: c.dir, consent: consentState(ctx, c), ...(m.login ? { login: !!loginOf(ctx.env, c) } : {}), direct, pace, routes: routesOf(c), via: routeOf(ctx.env, c).via },
|
|
242
|
+
};
|
|
243
|
+
},
|
|
244
|
+
}, {
|
|
245
|
+
path: ["connector", "use"],
|
|
246
|
+
summary: "Choose how a connector's images come: directly through strom, or through your own browser — switch any time",
|
|
247
|
+
group: "sources",
|
|
248
|
+
description: "Through the browser, the agent fetches them in your own browser (Claude in Chrome), where your login to the\n" +
|
|
249
|
+
"portal lives: strom still plans every request, paces it, and takes the files over from your downloads\n" +
|
|
250
|
+
"folder, checked and registered. The agent gets browser tools for the connector's sites only, from its next\n" +
|
|
251
|
+
"session on. Your agent may switch it when you ask it to.",
|
|
252
|
+
args: [{ name: "connector", description: "its name", required: true }],
|
|
253
|
+
options: [{ name: "via", type: "string", value: "<direct|browser>", description: "direct: strom fetches them · browser: your browser does" }],
|
|
254
|
+
examples: ["strom connector use example-archive --via browser", "strom connector use example-archive --via direct"],
|
|
255
|
+
run(ctx, { args, opts }) {
|
|
256
|
+
const c = findConnector(ctx.settings.shared()?.value, args[0]);
|
|
257
|
+
const via = String(opts.via ?? "").toLowerCase();
|
|
258
|
+
if (!ROUTES.includes(via))
|
|
259
|
+
throw new UsageError("--via direct or --via browser", { hint: `connector ${c.name} can: ${routesOf(c).join(", ")}` });
|
|
260
|
+
if (!routesOf(c).includes(via))
|
|
261
|
+
throw new UsageError(`connector ${c.name} cannot fetch ${via === "browser" ? "through the browser" : "directly — only through the browser"}`, {
|
|
262
|
+
hint: via === "browser" ? `its connector.json has no route "browser" (with "locate" in can) — whoever keeps it can add them: the contract, ${path.join(path.dirname(c.dir), "README.md")}` : `it needs the browser (a login the browser holds): strom connector use ${c.name} --via browser`,
|
|
263
|
+
});
|
|
264
|
+
if (via === "browser" && c.manifest.policy.automation === "manual")
|
|
265
|
+
throw new UsageError(`${c.manifest.title} does not allow automated download (its terms, as the connector read them) — through the browser neither`, { hint: 'the user saves the images by hand: strom task wait T… --images B…:<from-to> --on "<the book, its link, which images>"' });
|
|
266
|
+
const s = ctx.settings;
|
|
267
|
+
s.config.connectorRoutes = { ...(s.config.connectorRoutes ?? {}), [c.name]: { via, at: new Date().toISOString(), by: ctx.io.tty ? "in a terminal" : "by a command without a terminal (an agent or the app)" } };
|
|
268
|
+
s.save();
|
|
269
|
+
// the agents' permissions of every tree follow at once: browser tools for these sites, or none
|
|
270
|
+
const synced = [];
|
|
271
|
+
for (const known of ctx.knownTrees()) {
|
|
272
|
+
try {
|
|
273
|
+
const tree = Tree.open(known.root, ctx.env);
|
|
274
|
+
const files = syncAgentFiles(tree);
|
|
275
|
+
if (files.length) {
|
|
276
|
+
tree.withTreeLock(() => tree.commit(`Agent instructions: ${files.join(", ")}`, files));
|
|
277
|
+
synced.push(known.name);
|
|
278
|
+
}
|
|
279
|
+
}
|
|
280
|
+
catch {
|
|
281
|
+
// a tree strom cannot open now gets them at its next strom run
|
|
282
|
+
}
|
|
283
|
+
}
|
|
284
|
+
const hosts = c.manifest.hosts.map(bareHost);
|
|
285
|
+
return {
|
|
286
|
+
text: via === "browser"
|
|
287
|
+
? lines(`${c.name}: images through your browser from now on (${hosts.join(", ")})`, ` · the agent gets browser tools (Claude in Chrome) for these sites only, from its next session${synced.length ? ` (permissions of ${synced.join(", ")} updated)` : ""}. Chrome with the`, " Claude extension must be running then, signed in to the account Claude Code uses; browser tools need a model", " that can review its actions (Sonnet or Opus, not Haiku)", ` · strom plans and paces every request, and takes the files over from ${ctx.display(s.downloads())}`, " (another folder: strom config set browser.downloads <folder>)", ` · the first time, Chrome asks whether ${hosts[0]} may download several files: allow it once`, ` · back: strom connector use ${c.name} --via direct`)
|
|
288
|
+
: lines(`${c.name}: images directly through strom from now on, paced (${hosts.join(", ")})`, routesOf(c).includes("browser") ? ` · through your browser again: strom connector use ${c.name} --via browser` : undefined),
|
|
289
|
+
data: { connector: c.name, via, hosts, synced },
|
|
290
|
+
};
|
|
291
|
+
},
|
|
292
|
+
}, {
|
|
293
|
+
path: ["connector", "new"],
|
|
294
|
+
summary: "Start a connector for an archive portal in the plugins folder: the manifest, the SDK and the brief for your agent",
|
|
295
|
+
group: "sources",
|
|
296
|
+
description: "The brief (DISCOVERY.md) has the agent find out first what the portal allows — its terms of use,\n" +
|
|
297
|
+
"robots.txt, official exports — and only then write the connector against the contract (README.md\n" +
|
|
298
|
+
"of the plugins folder).",
|
|
299
|
+
args: [{ name: "name", description: "short name, lowercase — also its folder's name: state-archive", required: true }],
|
|
300
|
+
options: [
|
|
301
|
+
{ name: "url", type: "string", value: "<url>", description: "the portal (its start page) — required" },
|
|
302
|
+
{ name: "title", type: "string", value: "<text>", description: "the archive or portal, as people call it" },
|
|
303
|
+
],
|
|
304
|
+
examples: ['strom connector new state-archive --url https://archive.example.org --title "Example State Archive"'],
|
|
305
|
+
run(ctx, { args, opts }) {
|
|
306
|
+
const name = args[0];
|
|
307
|
+
if (!NAME_RE.test(name))
|
|
308
|
+
throw new UsageError("a connector's name is lowercase letters, digits and dashes", { hint: `e.g. ${suggestName(name) ?? "example-archive"}` });
|
|
309
|
+
if (!opts.url)
|
|
310
|
+
throw new UsageError("give the portal: --url https://…");
|
|
311
|
+
let url;
|
|
312
|
+
try {
|
|
313
|
+
url = new URL(String(opts.url));
|
|
314
|
+
}
|
|
315
|
+
catch {
|
|
316
|
+
throw new UsageError(`not a URL: ${opts.url}`);
|
|
317
|
+
}
|
|
318
|
+
const dir = path.join(ensurePluginsDir(shared(ctx)), name);
|
|
319
|
+
if (fs.existsSync(dir))
|
|
320
|
+
throw new UsageError(`${ctx.display(dir)} exists already`, { hint: `strom connector show ${name}` });
|
|
321
|
+
const title = String(opts.title ?? url.hostname);
|
|
322
|
+
const fill = (s) => s.replaceAll("__TITLE__", title).replaceAll("__URL__", url.origin).replaceAll("__NAME__", name);
|
|
323
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
324
|
+
const manifest = {
|
|
325
|
+
interface: INTERFACE,
|
|
326
|
+
title,
|
|
327
|
+
version: "0.1.0",
|
|
328
|
+
run: ["node", "connector.ts"],
|
|
329
|
+
hosts: [url.hostname],
|
|
330
|
+
can: ["find", "list", "fetch"],
|
|
331
|
+
policy: { automation: "unknown", terms: "", termsSummary: "", robots: "", officialExport: "" },
|
|
332
|
+
};
|
|
333
|
+
fs.writeFileSync(path.join(dir, MANIFEST), JSON.stringify(manifest, null, 2) + "\n");
|
|
334
|
+
fs.writeFileSync(path.join(dir, "package.json"), JSON.stringify({ type: "module", private: true }, null, 2) + "\n");
|
|
335
|
+
fs.writeFileSync(path.join(dir, "sdk.ts"), readAsset("plugins", "connectors", "sdk.ts"));
|
|
336
|
+
fs.writeFileSync(path.join(dir, "connector.ts"), fill(readAsset("plugins", "connectors", "template.ts")));
|
|
337
|
+
fs.writeFileSync(path.join(dir, "DISCOVERY.md"), fill(readAsset("plugins", "connectors", "DISCOVERY.md")));
|
|
338
|
+
fs.writeFileSync(path.join(dir, "README.md"), `# ${title} — strom connector\n\nPortal: ${url.origin}\n\nHow the portal is mapped (catalogue search, book IDs, image URLs, requests per image) — written by whoever builds it.\n`);
|
|
339
|
+
const rel = ctx.display(dir);
|
|
340
|
+
return {
|
|
341
|
+
text: lines(`connector ${name} started in ${rel}`, "", "next:", ` 1. your agent reads ${path.join(rel, "DISCOVERY.md")} and the contract (README.md one folder up),`, " and finds out what the portal allows — before any code", " 2. it writes connector.json (policy) and connector.ts, and tests with a few requests:", ` strom connector test ${name} --find "<place>"`, consentRequired(ctx.env) ? ` 3. the first test needs your consent, in your terminal: strom allow connector ${name}` : undefined),
|
|
342
|
+
data: { name, dir },
|
|
343
|
+
};
|
|
344
|
+
},
|
|
345
|
+
}, {
|
|
346
|
+
path: ["connector", "add"],
|
|
347
|
+
summary: "Install a connector from a folder or a git URL (you, in a terminal): copied into the plugins folder, then allowed",
|
|
348
|
+
group: "sources",
|
|
349
|
+
description: "The same as copying its folder into the plugins folder and allowing it: you read what it is and\n" +
|
|
350
|
+
"what it may contact first. Its name is the folder's name (or --name). An agent cannot run this.",
|
|
351
|
+
args: [{ name: "source", description: "folder or git URL", required: true }],
|
|
352
|
+
options: [{ name: "name", type: "string", value: "<name>", description: "its name here, when the folder's is not one (lowercase, digits, dashes)" }],
|
|
353
|
+
examples: ["strom connector add ~/Downloads/example-archive", "strom connector add https://github.com/someone/strom-connector-example.git --name example-archive"],
|
|
354
|
+
run: async (ctx, { args, opts }) => {
|
|
355
|
+
const source = args[0];
|
|
356
|
+
ctx.requireHuman(`Install and allow the connector from ${source}?`, `strom connector add ${shellArg(source)}`, "connector", undefined, { window: false });
|
|
357
|
+
let dir = path.resolve(ctx.cwd, source);
|
|
358
|
+
let tmp;
|
|
359
|
+
try {
|
|
360
|
+
if (!fs.existsSync(dir)) {
|
|
361
|
+
if (!/^(https?:\/\/|git@|ssh:\/\/)/.test(source))
|
|
362
|
+
throw new UsageError(`no folder ${source}`, { hint: "a folder with connector.json, or a git URL" });
|
|
363
|
+
tmp = fs.mkdtempSync(path.join(os.tmpdir(), "strom-connector-"));
|
|
364
|
+
const r = runGit(tmp, ["clone", "--depth", "1", source, "c"]);
|
|
365
|
+
if (r.status !== 0)
|
|
366
|
+
throw new StromError(`git clone failed: ${r.stderr.trim().split("\n").at(-1)}`);
|
|
367
|
+
dir = path.join(tmp, "c");
|
|
368
|
+
}
|
|
369
|
+
const m = readManifest(dir);
|
|
370
|
+
const base = path.basename(tmp ? source.replace(/\/+$/, "").replace(/\.git$/, "") : dir).replace(/^strom-connector-/, "");
|
|
371
|
+
const name = String(opts.name ?? (NAME_RE.test(base) ? base : (suggestName(base) ?? "")));
|
|
372
|
+
if (!NAME_RE.test(name))
|
|
373
|
+
throw new UsageError(`"${opts.name ?? base}" is not a connector's name: lowercase letters, digits and dashes`, { hint: `strom connector add ${shellArg(source)} --name <name>` });
|
|
374
|
+
const target = path.join(ensurePluginsDir(shared(ctx)), name);
|
|
375
|
+
const same = path.resolve(dir) === path.resolve(target);
|
|
376
|
+
ctx.io.stdout(codeWarning({ name, dir: target, manifest: m }, source, directNetwork({ dir, manifest: m })) + "\n");
|
|
377
|
+
if (!same && fs.existsSync(target))
|
|
378
|
+
ctx.io.stdout(` ⚠ it replaces the connector ${name} that is there now\n`);
|
|
379
|
+
if (!(await ctx.confirm(`Install and allow connector ${name}?`, false)))
|
|
380
|
+
return { text: "not allowed — nothing changed", data: { allowed: false } };
|
|
381
|
+
if (!same) {
|
|
382
|
+
fs.rmSync(target, { recursive: true, force: true });
|
|
383
|
+
fs.cpSync(dir, target, { recursive: true, filter: (f) => ![".git", ".test"].includes(path.basename(f)) });
|
|
384
|
+
}
|
|
385
|
+
const c = { name, dir: target, manifest: m };
|
|
386
|
+
grant(ctx, c, source);
|
|
387
|
+
const r = await askConnector(ctx, c, source);
|
|
388
|
+
return {
|
|
389
|
+
text: lines(`connector ${name} allowed${same ? "" : `, installed in ${ctx.display(target)}`}`, r.missing.length ? `hosts not allowed: ${r.missing.join(", ")} — it cannot reach them (strom allow connector ${name})` : `hosts allowed: ${r.hosts.join(", ")}`, m.can.includes("find") ? `find books: strom fetch ${name} --find "<place>" --years <from-to>` : undefined),
|
|
390
|
+
data: { allowed: true, name, dir: target, hosts: r.hosts, missing: r.missing },
|
|
391
|
+
};
|
|
392
|
+
}
|
|
393
|
+
finally {
|
|
394
|
+
if (tmp)
|
|
395
|
+
fs.rmSync(tmp, { recursive: true, force: true });
|
|
396
|
+
}
|
|
397
|
+
},
|
|
398
|
+
}, {
|
|
399
|
+
path: ["connector", "remove"],
|
|
400
|
+
summary: "Remove a connector (you, in a terminal): its folder is deleted and its consent goes",
|
|
401
|
+
group: "sources",
|
|
402
|
+
args: [{ name: "connector", description: "its name", required: true }],
|
|
403
|
+
examples: ["strom connector remove example-archive"],
|
|
404
|
+
run: async (ctx, { args }) => {
|
|
405
|
+
const name = args[0];
|
|
406
|
+
ctx.requireHuman(`Remove connector ${name}?`, `strom connector remove ${name}`, "connector", undefined, { window: false });
|
|
407
|
+
const dir = path.join(connectorsDir(shared(ctx)), name);
|
|
408
|
+
if (!NAME_RE.test(name) || !fs.existsSync(dir))
|
|
409
|
+
throw new UsageError(`no connector "${name}"`, { hint: "strom connector list" });
|
|
410
|
+
if (!(await ctx.confirm(`Remove connector ${name} (deletes ${ctx.display(dir)})?`, false)))
|
|
411
|
+
return { text: "nothing changed" };
|
|
412
|
+
const all = loadConsents(ctx.env);
|
|
413
|
+
delete all.connectors[name];
|
|
414
|
+
saveConsents(ctx.env, all);
|
|
415
|
+
const forgot = removeLogin(ctx.env, name);
|
|
416
|
+
fs.rmSync(dir, { recursive: true, force: true });
|
|
417
|
+
return { text: `connector ${name} removed${forgot ? ", and your login for it forgotten" : ""} — hosts you allowed stay allowed (strom consents)`, data: { removed: name } };
|
|
418
|
+
},
|
|
419
|
+
}, {
|
|
420
|
+
path: ["connector", "test"],
|
|
421
|
+
summary: "Try a connector with a few requests (at most 10): what it finds, lists or fetches — files into its .test folder",
|
|
422
|
+
group: "sources",
|
|
423
|
+
args: [{ name: "connector", description: "its name", required: true }],
|
|
424
|
+
options: [
|
|
425
|
+
{ name: "find", type: "string", value: "<place>", description: "find the books of a place" },
|
|
426
|
+
{ name: "years", type: "string", value: "<from-to>", description: "with --find" },
|
|
427
|
+
{ name: "list", type: "string", value: "<book>", description: "describe one book" },
|
|
428
|
+
{ name: "fetch", type: "string", value: "<book>", description: "fetch images of a book (with --images)" },
|
|
429
|
+
{ name: "locate", type: "string", value: "<book>", description: "where images of a book are, for the browser (with --images)" },
|
|
430
|
+
{ name: "images", type: "string", value: "<from-to>", description: "which images (default 1)" },
|
|
431
|
+
{ name: "crop", type: "string", value: "<x,y,w,h>", description: "with --fetch: a part of one image, in fractions of it" },
|
|
432
|
+
{ name: "half", type: "string", value: "<side>", description: "with --fetch: a half of one image (left, right, top, bottom)" },
|
|
433
|
+
{ name: "max", type: "string", value: "<n>", description: "requests at most (default 10)" },
|
|
434
|
+
],
|
|
435
|
+
examples: [
|
|
436
|
+
'strom connector test example-archive --find "Týnec" --years 1780-1850',
|
|
437
|
+
"strom connector test example-archive --fetch 5359 --images 1-2",
|
|
438
|
+
"strom connector test example-archive --fetch 5359 --images 2 --crop 0.5,0,0.5,0.5",
|
|
439
|
+
],
|
|
440
|
+
run: async (ctx, { args, opts }) => {
|
|
441
|
+
const c = findConnector(ctx.settings.shared()?.value, args[0]);
|
|
442
|
+
const request = requestOf(opts, true);
|
|
443
|
+
const max = opts.max === undefined ? 10 : Number(opts.max);
|
|
444
|
+
if (!Number.isInteger(max) || max < 1 || max > 50)
|
|
445
|
+
throw new UsageError("--max: 1 to 50");
|
|
446
|
+
return testWith(ctx, c, request, max);
|
|
447
|
+
},
|
|
448
|
+
}, {
|
|
449
|
+
path: ["connector", "probe"],
|
|
450
|
+
summary: "One request through strom while you build a connector: the answer saved in its .test/probe folder, to read",
|
|
451
|
+
group: "sources",
|
|
452
|
+
description: "For mapping a portal whose pages are built by JavaScript: fetch its page, then the script it loads, and\n" +
|
|
453
|
+
"look in them for the addresses its search and viewer use. Paced by strom and only to the connector's hosts,\n" +
|
|
454
|
+
"like the connector itself. Each probe is one request: keep them few. Probes keep the cookies servers set,\n" +
|
|
455
|
+
"like one visit in a browser (a session, the result of a form); --fresh starts without them.",
|
|
456
|
+
args: [
|
|
457
|
+
{ name: "connector", description: "its name", required: true },
|
|
458
|
+
{ name: "url", description: "on one of its hosts", required: true },
|
|
459
|
+
],
|
|
460
|
+
options: [
|
|
461
|
+
{ name: "save", type: "string", value: "<file>", description: "file name in .test/probe (default: from the URL)" },
|
|
462
|
+
{ name: "method", type: "string", value: "<GET|POST|HEAD>", description: "default GET" },
|
|
463
|
+
{ name: "body", type: "string", value: "<text>", description: "the body of a POST" },
|
|
464
|
+
{ name: "header", type: "string", multiple: true, value: "<Name: value>", description: "a request header (repeatable)" },
|
|
465
|
+
{ name: "fresh", type: "boolean", description: "start without the cookies of earlier probes (a new session)" },
|
|
466
|
+
],
|
|
467
|
+
examples: [
|
|
468
|
+
"strom connector probe example-archive https://archive.example.org/",
|
|
469
|
+
'strom connector probe example-archive https://archive.example.org/api/search --method POST --body "{\\"q\\":\\"Týnec\\"}" --header "Content-Type: application/json"',
|
|
470
|
+
],
|
|
471
|
+
run: async (ctx, { args, opts }) => {
|
|
472
|
+
const c = findConnector(ctx.settings.shared()?.value, args[0]);
|
|
473
|
+
const url = args[1];
|
|
474
|
+
const method = String(opts.method ?? "GET").toUpperCase();
|
|
475
|
+
if (!["GET", "POST", "HEAD"].includes(method))
|
|
476
|
+
throw new UsageError("--method: GET, POST or HEAD");
|
|
477
|
+
const headers = {};
|
|
478
|
+
for (const h of opts.header ?? []) {
|
|
479
|
+
const i = h.indexOf(":");
|
|
480
|
+
if (i <= 0)
|
|
481
|
+
throw new UsageError(`--header "Name: value", not "${h}"`);
|
|
482
|
+
headers[h.slice(0, i).trim()] = h.slice(i + 1).trim();
|
|
483
|
+
}
|
|
484
|
+
if (!(await ensureAllowed(ctx, c)))
|
|
485
|
+
return { text: "not allowed — nothing was sent", exitCode: 1 };
|
|
486
|
+
const dir = path.join(c.dir, ".test", "probe");
|
|
487
|
+
if (c.manifest.browser?.pages) {
|
|
488
|
+
// the portal answers a real browser only: the probe is a page of the user's browser too
|
|
489
|
+
const want = { method: method, url, headers, ...(opts.body !== undefined ? { body: String(opts.body) } : {}) };
|
|
490
|
+
const u = URL.canParse(url) ? new URL(url) : undefined;
|
|
491
|
+
if (!u || !hostAllowed(u.hostname, c.manifest.hosts))
|
|
492
|
+
throw new UsageError(`${u?.hostname ?? url} is not one of its hosts (${c.manifest.hosts.join(", ")})`);
|
|
493
|
+
const save = opts.save === undefined ? undefined : String(opts.save);
|
|
494
|
+
const got = loadPage(ctx.tree().root, c.name, pageKey(want));
|
|
495
|
+
if (got)
|
|
496
|
+
return probeResult(ctx, c, method, url, got, save);
|
|
497
|
+
return planPages(ctx, c, want, { cmd: "probe", ...(save ? { save } : {}) }, `probe ${url}`);
|
|
498
|
+
}
|
|
499
|
+
const jarFile = path.join(dir, ".cookies.json");
|
|
500
|
+
let kept = [];
|
|
501
|
+
try {
|
|
502
|
+
if (!opts.fresh)
|
|
503
|
+
kept = JSON.parse(fs.readFileSync(jarFile, "utf8"));
|
|
504
|
+
}
|
|
505
|
+
catch { }
|
|
506
|
+
const cookies = CookieJar.from(kept);
|
|
507
|
+
const res = await politeRequest(url, {
|
|
508
|
+
stateDir: netDir(ctx),
|
|
509
|
+
hosts: c.manifest.hosts,
|
|
510
|
+
pace: c.manifest.policy.pace ?? {},
|
|
511
|
+
method: method,
|
|
512
|
+
headers,
|
|
513
|
+
...(opts.body !== undefined ? { body: String(opts.body) } : {}),
|
|
514
|
+
cookies,
|
|
515
|
+
});
|
|
516
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
517
|
+
fs.writeFileSync(jarFile, JSON.stringify(cookies.entries(), null, 2) + "\n");
|
|
518
|
+
return probeResult(ctx, c, method, url, res, opts.save === undefined ? undefined : String(opts.save), [...new Set(cookies.entries().map((k) => k.name))]);
|
|
519
|
+
},
|
|
520
|
+
}, {
|
|
521
|
+
path: ["connector", "grep"],
|
|
522
|
+
summary: "Search what its probes and tests saved (addresses, form fields, scripts): each hit with the text round it",
|
|
523
|
+
group: "sources",
|
|
524
|
+
description: "The pages and scripts of a portal are often too big to read whole, and many are one long line. This finds\n" +
|
|
525
|
+
"a text (in any case), or a regular expression with --regex, in the files of the connector's .test folder\n" +
|
|
526
|
+
"and shows each hit with a little text round it and where it is (file:line:column). Nothing is requested.\n" +
|
|
527
|
+
"It finds a text as it reads: á or á of a page and \\u00e1 of a script are á, is a space.\n" +
|
|
528
|
+
"The text round a hit is shown as the file has it, which is what a connector reads.",
|
|
529
|
+
args: [
|
|
530
|
+
{ name: "connector", description: "its name", required: true },
|
|
531
|
+
{ name: "pattern", description: "the text to find: in any case, as it reads (á is á), a space for any white space", required: true },
|
|
532
|
+
],
|
|
533
|
+
options: [
|
|
534
|
+
{ name: "regex", type: "boolean", description: "the pattern is a regular expression (JavaScript, in any case)" },
|
|
535
|
+
{ name: "in", type: "string", value: "<file>", description: "only this file (its name in .test/probe, or its path in .test)" },
|
|
536
|
+
{ name: "around", type: "string", value: "<n>", description: "characters shown before and after a hit (default 80)" },
|
|
537
|
+
],
|
|
538
|
+
examples: [
|
|
539
|
+
'strom connector grep example-archive "/api/"',
|
|
540
|
+
'strom connector grep example-archive "\\.jpe?g" --regex --in viewer.html',
|
|
541
|
+
],
|
|
542
|
+
run: (ctx, { args, opts }) => {
|
|
543
|
+
const c = findConnector(ctx.settings.shared()?.value, args[0]);
|
|
544
|
+
const pattern = args[1];
|
|
545
|
+
const around = opts.around === undefined ? 80 : Number(opts.around);
|
|
546
|
+
if (!Number.isInteger(around) || around < 0 || around > 2000)
|
|
547
|
+
throw new UsageError("--around: a number of characters, 0–2000");
|
|
548
|
+
const re = grepPattern(pattern, !!opts.regex);
|
|
549
|
+
const base = path.join(c.dir, ".test");
|
|
550
|
+
let files = textFiles(base);
|
|
551
|
+
if (opts.in !== undefined) {
|
|
552
|
+
const want = String(opts.in);
|
|
553
|
+
files = files.filter((f) => [path.relative(base, f), path.relative(path.join(base, "probe"), f)].some((r) => r.split(path.sep).join("/") === want.split(path.sep).join("/")));
|
|
554
|
+
if (!files.length)
|
|
555
|
+
throw new UsageError(`no saved file ${want} in ${ctx.display(base)}`, { hint: `strom connector grep ${c.name} ${shellArg(pattern)}${opts.regex ? " --regex" : ""} — to search them all` });
|
|
556
|
+
}
|
|
557
|
+
if (!files.length)
|
|
558
|
+
return { text: `nothing saved yet — strom connector probe ${c.name} <url> saves an answer to search`, data: { total: 0, hits: [] } };
|
|
559
|
+
// hits close together (a long list of names) are shown once, in one stretch of text
|
|
560
|
+
const hits = [];
|
|
561
|
+
let total = 0;
|
|
562
|
+
for (const f of files) {
|
|
563
|
+
const text = fs.readFileSync(f, "utf8");
|
|
564
|
+
// found as the file has it, and as it reads (á or \u00e1 written out): each hit once
|
|
565
|
+
const found = new Map();
|
|
566
|
+
for (const m of text.matchAll(re)) {
|
|
567
|
+
if (found.size >= 5000)
|
|
568
|
+
break;
|
|
569
|
+
if (m[0])
|
|
570
|
+
found.set(m.index, m.index + m[0].length);
|
|
571
|
+
}
|
|
572
|
+
const read = readable(text);
|
|
573
|
+
for (const m of read ? read.text.matchAll(re) : []) {
|
|
574
|
+
if (found.size >= 5000)
|
|
575
|
+
break;
|
|
576
|
+
const at = read.at[m.index];
|
|
577
|
+
if (m[0] && !found.has(at))
|
|
578
|
+
found.set(at, read.at[m.index + m[0].length]);
|
|
579
|
+
}
|
|
580
|
+
let open;
|
|
581
|
+
const close = () => {
|
|
582
|
+
if (!open)
|
|
583
|
+
return;
|
|
584
|
+
const lineStart = text.lastIndexOf("\n", open.at - 1) + 1;
|
|
585
|
+
hits.push({
|
|
586
|
+
file: path.relative(base, f).split(path.sep).join("/"),
|
|
587
|
+
line: text.slice(0, open.at).split("\n").length,
|
|
588
|
+
column: open.at - lineStart + 1,
|
|
589
|
+
hits: open.n,
|
|
590
|
+
text: (open.from > 0 ? "…" : "") + text.slice(open.from, open.to).replace(/\s+/g, " ") + (open.to < text.length ? "…" : ""),
|
|
591
|
+
});
|
|
592
|
+
open = undefined;
|
|
593
|
+
};
|
|
594
|
+
for (const [at, end] of [...found].sort((x, y) => x[0] - y[0])) {
|
|
595
|
+
total++;
|
|
596
|
+
const from = Math.max(0, at - around);
|
|
597
|
+
const to = Math.min(text.length, end + around);
|
|
598
|
+
if (open && from <= open.to && to - open.from <= Math.max(600, 4 * around)) {
|
|
599
|
+
open.to = to;
|
|
600
|
+
open.n++;
|
|
601
|
+
}
|
|
602
|
+
else {
|
|
603
|
+
close();
|
|
604
|
+
open = { from, to, at, n: 1 };
|
|
605
|
+
}
|
|
606
|
+
if (total >= 5000)
|
|
607
|
+
break;
|
|
608
|
+
}
|
|
609
|
+
close();
|
|
610
|
+
if (total >= 5000)
|
|
611
|
+
break;
|
|
612
|
+
}
|
|
613
|
+
const page = paginate(hits, ctx.limit, ctx.page);
|
|
614
|
+
const inFiles = new Set(hits.map((h) => h.file)).size;
|
|
615
|
+
return {
|
|
616
|
+
text: hits.length
|
|
617
|
+
? lines(`${total}${total >= 5000 ? "+" : ""} hit(s) in ${inFiles} file(s) of ${ctx.display(base)}:`, ...page.items.map((h) => `${h.file}:${h.line}:${h.column} ${h.hits > 1 ? `(${h.hits} hits) ` : ""}${h.text}`), moreLine(page, `strom connector grep ${c.name} ${shellArg(pattern)}${opts.regex ? " --regex" : ""}${opts.in !== undefined ? ` --in ${shellArg(String(opts.in))}` : ""}`) || undefined)
|
|
618
|
+
: `not found in ${files.length} saved file(s) of ${ctx.display(base)}`,
|
|
619
|
+
data: { total, hits: page.items },
|
|
620
|
+
};
|
|
621
|
+
},
|
|
622
|
+
}, {
|
|
623
|
+
path: ["fetch"],
|
|
624
|
+
summary: "Download through a connector: images of a book (registered with their provenance), or find books of a place",
|
|
625
|
+
group: "sources",
|
|
626
|
+
tree: true,
|
|
627
|
+
writes: true,
|
|
628
|
+
lock: "sections",
|
|
629
|
+
description: "Every request goes through strom's limiter: one at a time, paced, capped per hour; a refusal (401/403)\n" +
|
|
630
|
+
"stops the run and leaves the archive alone for a day. With --recordset the images are registered at once;\n" +
|
|
631
|
+
"without it they go into the inbox, a folder for the book.\n" +
|
|
632
|
+
"--crop or --half fetches a part of one image, as sharp as the portal gives it (a connector that can: part),\n" +
|
|
633
|
+
"registered with the image; a view of the image then shows that place from it by itself. The book is\n" +
|
|
634
|
+
"known from images the connector fetched for the record set before.\n" +
|
|
635
|
+
"A connector set to fetch through your browser (strom connector use <c> --via browser) gives a plan instead:\n" +
|
|
636
|
+
"the page to open, a script to run there that fetches the images at the archive's pace into the browser's\n" +
|
|
637
|
+
"downloads folder, and --take, which takes them over from there, checked and registered.\n" +
|
|
638
|
+
"A connector whose portal answers a real browser only (browser.pages: a bot check) has every page read\n" +
|
|
639
|
+
"there too: each run plans the page it needs next, --take gives it to the connector and goes on.",
|
|
640
|
+
args: [
|
|
641
|
+
{ name: "connector", description: "its name", required: true },
|
|
642
|
+
{ name: "book", description: "the book, as the connector knows it (its ID on the portal)" },
|
|
643
|
+
],
|
|
644
|
+
options: [
|
|
645
|
+
{ name: "images", type: "string", value: "<from-to>", description: "which images of the book" },
|
|
646
|
+
{ name: "recordset", type: "string", value: "<B…>", description: "register them as images of this record set" },
|
|
647
|
+
{ name: "crop", type: "string", value: "<x,y,w,h>", description: "a part of one image: fractions of it (0.5,0.2,0.5,0.3), or pixels of the registered image" },
|
|
648
|
+
{ name: "half", type: "string", value: "<side>", description: "a half of one image: left, right, top or bottom (with --crop: within it)" },
|
|
649
|
+
{ name: "find", type: "string", value: "<place>", description: "find the books that cover a place" },
|
|
650
|
+
{ name: "years", type: "string", value: "<from-to>", description: "with --find" },
|
|
651
|
+
{ name: "list", type: "boolean", description: "describe the book (title, images)" },
|
|
652
|
+
{ name: "take", type: "boolean", description: "through the browser: take over what it downloaded (checked, registered)" },
|
|
653
|
+
{ name: "result", type: "string", value: "<line>", description: "with --take: the line the browser script returned (strom-result …)" },
|
|
654
|
+
],
|
|
655
|
+
examples: [
|
|
656
|
+
'strom fetch example-archive --find "Týnec" --years 1780-1850',
|
|
657
|
+
"strom fetch example-archive 5359 --images 40-69 --recordset B0001",
|
|
658
|
+
"strom fetch example-archive 5359 --recordset B0001 --images 2 --crop 0.5,0.4,0.5,0.3",
|
|
659
|
+
'strom fetch example-archive --take --result "strom-result 40:200:1843221 41:200:1790234"',
|
|
660
|
+
],
|
|
661
|
+
run: async (ctx, { args, opts }) => {
|
|
662
|
+
const tree = ctx.tree();
|
|
663
|
+
const c = findConnector(ctx.settings.shared()?.value, args[0]);
|
|
664
|
+
if (opts.take)
|
|
665
|
+
return takeOver(ctx, c, opts.result === undefined ? undefined : String(opts.result));
|
|
666
|
+
const perHour = paceOf(c.manifest.policy.pace).perHour;
|
|
667
|
+
const wantsPart = opts.crop !== undefined || opts.half !== undefined;
|
|
668
|
+
if (wantsPart && !c.manifest.can.includes("part"))
|
|
669
|
+
throw new UsageError(`connector ${c.name} fetches whole images only (it cannot: part)`, {
|
|
670
|
+
hint: 'the user saves the part by hand: zoomed in on it in the portal\'s viewer — strom task wait T… --images B…:<n> --on "<the book, its link, the image, the entry to zoom in on; saved as 40a.jpg>"',
|
|
671
|
+
});
|
|
672
|
+
const request = opts.find
|
|
673
|
+
? requestOf(opts, false)
|
|
674
|
+
: opts.list
|
|
675
|
+
? { cmd: "list", book: need(args[1], "the book") }
|
|
676
|
+
: wantsPart
|
|
677
|
+
? partRequest(tree, c, args[1], opts)
|
|
678
|
+
: { cmd: "fetch", book: need(args[1], "the book"), images: parseImages(opts.images, perHour) ?? needImages() };
|
|
679
|
+
const recordset = opts.recordset ? requireRecord(tree, String(opts.recordset), "recordset").id : undefined;
|
|
680
|
+
return fetchWith(ctx, c, request, recordset);
|
|
681
|
+
},
|
|
682
|
+
}, {
|
|
683
|
+
path: ["allow", "connector"],
|
|
684
|
+
summary: "Allow a connector to run (you, in a terminal, after reading what it is and what it contacts) — or take it back",
|
|
685
|
+
group: "setup",
|
|
686
|
+
description: "Allows its folder and automated access to each of its hosts. While its code reaches the network only\n" +
|
|
687
|
+
"through strom, your agent may go on improving it; code that goes round strom needs a yes after every change.",
|
|
688
|
+
args: [{ name: "connector", description: "its name", required: true }],
|
|
689
|
+
options: [{ name: "revoke", type: "boolean", description: "take the consent back (its hosts stay allowed: strom allow host <host> --revoke)" }],
|
|
690
|
+
examples: ["strom allow connector example-archive", "strom allow connector example-archive --revoke"],
|
|
691
|
+
run: async (ctx, { args, opts }) => {
|
|
692
|
+
const name = args[0];
|
|
693
|
+
if (!opts.revoke && !consentRequired(ctx.env)) {
|
|
694
|
+
const c = findConnector(shared(ctx), name);
|
|
695
|
+
if (!directNetwork(c).length)
|
|
696
|
+
return {
|
|
697
|
+
text: `consents are off: connector ${name} runs without asking — paced by strom, only to ${c.manifest.hosts.join(", ")}\nto be asked first: strom config set connectors.consent on`,
|
|
698
|
+
data: { name, allowed: true, consents: "off" },
|
|
699
|
+
};
|
|
700
|
+
}
|
|
701
|
+
const how = ctx.requireHuman(`${opts.revoke ? "Take back" : "Allow"} connector ${name}?`, `strom allow connector ${name}${opts.revoke ? " --revoke" : ""}`, `connector:${name}`, opts.revoke ? ui(ctx.uiLang(), "ui.consent.connector.revoke", { name }) : consentWindow(ctx, findConnector(shared(ctx), name)));
|
|
702
|
+
if (opts.revoke) {
|
|
703
|
+
const all = loadConsents(ctx.env);
|
|
704
|
+
const had = !!all.connectors[name];
|
|
705
|
+
delete all.connectors[name];
|
|
706
|
+
saveConsents(ctx.env, all);
|
|
707
|
+
return { text: had ? `connector ${name}: consent taken back — it does not run now` : `connector ${name} was not allowed — nothing changed`, data: { name, allowed: false } };
|
|
708
|
+
}
|
|
709
|
+
const c = findConnector(shared(ctx), name);
|
|
710
|
+
const before = missingConsents(ctx.env, c);
|
|
711
|
+
if (!before.code && !before.hosts.length)
|
|
712
|
+
return { text: `connector ${name} is allowed already, with its hosts ${c.manifest.hosts.map(bareHost).join(", ")}`, data: { name, allowed: true } };
|
|
713
|
+
const r = await askConnector(ctx, c, ctx.display(c.dir), how === "window");
|
|
714
|
+
if (!r.allowed)
|
|
715
|
+
return { text: "not allowed — nothing changed", data: { name, allowed: false } };
|
|
716
|
+
return {
|
|
717
|
+
text: lines(`connector ${name} allowed`, r.missing.length ? `hosts not allowed: ${r.missing.join(", ")} — it cannot reach them` : `hosts allowed: ${r.hosts.join(", ")}`),
|
|
718
|
+
data: { name, allowed: true, hosts: r.hosts, missing: r.missing },
|
|
719
|
+
};
|
|
720
|
+
},
|
|
721
|
+
}, {
|
|
722
|
+
path: ["allow", "host"],
|
|
723
|
+
summary: "Allow automated access to an archive's host (you, in a terminal, after the warning) — or take it back",
|
|
724
|
+
group: "setup",
|
|
725
|
+
args: [{ name: "host", description: "e.g. digi.example.org", required: true }],
|
|
726
|
+
options: [
|
|
727
|
+
{ name: "revoke", type: "boolean", description: "take the consent back" },
|
|
728
|
+
{ name: "unblock", type: "boolean", description: "lift a refusal (401/403) early — only after the archive said it is fine" },
|
|
729
|
+
],
|
|
730
|
+
examples: ["strom allow host archive.example.org", "strom allow host archive.example.org --revoke"],
|
|
731
|
+
run: async (ctx, { args, opts }) => {
|
|
732
|
+
const host = args[0].toLowerCase().replace(/^https?:\/\//, "").replace(/\/.*$/, "");
|
|
733
|
+
const lang = ctx.uiLang();
|
|
734
|
+
const how = ctx.requireHuman(`${opts.revoke ? "Take back" : opts.unblock ? "Lift the refusal of" : "Allow"} automated access to ${host}?`, `strom allow host ${host}${opts.revoke ? " --revoke" : opts.unblock ? " --unblock" : ""}`, `host:${host}`, ui(lang, opts.revoke ? "ui.consent.host.revoke" : opts.unblock ? "ui.consent.host.unblock" : "ui.consent.host", { host }));
|
|
735
|
+
const all = loadConsents(ctx.env);
|
|
736
|
+
if (opts.revoke) {
|
|
737
|
+
delete all.hosts[host];
|
|
738
|
+
saveConsents(ctx.env, all);
|
|
739
|
+
return { text: `${host}: consent taken back — no connector reaches it now`, data: { host, allowed: false } };
|
|
740
|
+
}
|
|
741
|
+
if (opts.unblock) {
|
|
742
|
+
if (how === "terminal" && !(await ctx.confirm(`${host} refused us. Lift it now (only if the archive said it is fine)?`, false)))
|
|
743
|
+
return { text: "nothing changed" };
|
|
744
|
+
clearBlock(netDir(ctx), host);
|
|
745
|
+
return { text: `${host}: the refusal is lifted — strom will ask it again, slowly`, data: { host, unblocked: true } };
|
|
746
|
+
}
|
|
747
|
+
const c = listConnectors(ctx.settings.shared()?.value).find((x) => x.manifest.hosts.some((h) => hostAllowed(host, [h])));
|
|
748
|
+
const ok = await askHost(ctx, host, c, how === "window");
|
|
749
|
+
return { text: ok ? `${host}: automated access allowed (strom consents)` : "not allowed — nothing changed", data: { host, allowed: ok } };
|
|
750
|
+
},
|
|
751
|
+
}, {
|
|
752
|
+
path: ["consents"],
|
|
753
|
+
summary: "Consents (on or off, what you allowed) and archive hosts: how each is doing (paced, capped, refused)",
|
|
754
|
+
group: "setup",
|
|
755
|
+
run(ctx) {
|
|
756
|
+
const all = loadConsents(ctx.env);
|
|
757
|
+
const hosts = [...new Set([...Object.keys(all.hosts), ...knownHosts(netDir(ctx))])].sort();
|
|
758
|
+
const names = Object.keys(all.connectors).sort();
|
|
759
|
+
return {
|
|
760
|
+
text: lines(consentRequired(ctx.env)
|
|
761
|
+
? "consents: on — a connector runs only once you allow it, and each archive host it contacts"
|
|
762
|
+
: "consents: off — connectors run without asking (paced by strom, only to their hosts); code that goes round strom needs your yes · to be asked first: strom config set connectors.consent on", "connectors", ...names.map((n) => ` ${n} ${ctx.display(all.connectors[n].dir)} since ${all.connectors[n].at.slice(0, 10)} hosts ${all.connectors[n].hosts.join(", ")}`), names.length ? undefined : " (none)", "hosts (automated access)", ...hosts.map((h) => ` ${hostLine(ctx, h)}${all.hosts[h] ? ` · allowed since ${all.hosts[h].at.slice(0, 10)}` : ""}`), hosts.length ? undefined : " (none)", `pace: one request at a time per host, ≥${DEFAULT_PACE.minIntervalMs / 1000} s apart, ≤${DEFAULT_PACE.perHour}/h; a refusal stops for a day`, Object.keys(loadLogins(ctx.env)).length ? `logins saved (strom login): ${Object.keys(loadLogins(ctx.env)).sort().join(", ")}` : undefined, "take back: strom allow connector <name> --revoke · strom allow host <host> --revoke"),
|
|
763
|
+
data: { consents: consentRequired(ctx.env) ? "on" : "off", ...all, logins: Object.keys(loadLogins(ctx.env)).sort(), hostState: Object.fromEntries(hosts.map((h) => [h, hostState(netDir(ctx), h)])) },
|
|
764
|
+
};
|
|
765
|
+
},
|
|
766
|
+
}, {
|
|
767
|
+
path: ["login"],
|
|
768
|
+
summary: "Your login to an archive portal, for a connector that can use one (you, in a terminal) — kept on this computer only",
|
|
769
|
+
group: "setup",
|
|
770
|
+
description: "An account you have on a portal — perhaps one you paid for — lets its connector fetch what the account\n" +
|
|
771
|
+
"gives. You type it in here; an agent cannot. It is kept in your config folder, readable by you alone,\n" +
|
|
772
|
+
"never in a tree, the shared folder or git. strom puts it into the connector's requests itself, only to\n" +
|
|
773
|
+
"the connector's hosts and only over https, and takes it out of every answer: the connector's code never\n" +
|
|
774
|
+
"sees it, nor does your agent. Without a connector: the logins saved.",
|
|
775
|
+
args: [{ name: "connector", description: "its name" }],
|
|
776
|
+
options: [{ name: "remove", type: "boolean", description: "forget the login" }],
|
|
777
|
+
examples: ["strom login example-archive", "strom login example-archive --remove", "strom login"],
|
|
778
|
+
run: async (ctx, { args, opts }) => {
|
|
779
|
+
const saved = loadLogins(ctx.env);
|
|
780
|
+
if (!args[0]) {
|
|
781
|
+
const usable = listConnectors(ctx.settings.shared()?.value).filter((c) => c.manifest.login);
|
|
782
|
+
const names = [...new Set([...Object.keys(saved), ...usable.map((c) => c.name)])].sort();
|
|
783
|
+
return {
|
|
784
|
+
text: names.length
|
|
785
|
+
? lines(...names.map((n) => {
|
|
786
|
+
const c = usable.find((x) => x.name === n);
|
|
787
|
+
return `${n} ${saved[n] ? `saved ${saved[n].at.slice(0, 10)} (${fieldsOf(saved[n].values, c).join(", ")})` : "none"}${c ? ` — ${c.manifest.login.about}` : " — no such connector now"}`;
|
|
788
|
+
}), "save one: strom login <connector> · forget it: strom login <connector> --remove")
|
|
789
|
+
: "no connector here uses a login",
|
|
790
|
+
data: { logins: names.map((n) => ({ connector: n, saved: !!saved[n], at: saved[n]?.at, fields: saved[n] ? fieldsOf(saved[n].values, usable.find((x) => x.name === n)) : [] })) },
|
|
791
|
+
};
|
|
792
|
+
}
|
|
793
|
+
const name = args[0];
|
|
794
|
+
if (opts.remove) {
|
|
795
|
+
ctx.requireHuman(`Forget your login for connector ${name}?`, `strom login ${name} --remove`, `login:${name}`, undefined, { window: false });
|
|
796
|
+
return removeLogin(ctx.env, name)
|
|
797
|
+
? { text: `the login for ${name} is forgotten — deleted from this computer`, data: { connector: name, saved: false } }
|
|
798
|
+
: { text: `no login saved for ${name} — nothing changed`, data: { connector: name, saved: false } };
|
|
799
|
+
}
|
|
800
|
+
const c = findConnector(shared(ctx), name);
|
|
801
|
+
if (!c.manifest.login)
|
|
802
|
+
throw new UsageError(`connector ${name} uses no login`, { hint: `a connector that can use one says so in its connector.json ("login"): ${path.join(connectorsDir(shared(ctx)), "README.md")}` });
|
|
803
|
+
ctx.requireHuman(`Save your login to ${c.manifest.title} for connector ${name}?`, `strom login ${name}`, `login:${name}`, undefined, { window: false });
|
|
804
|
+
if (loginOf(ctx.env, c) && !(await ctx.confirm(`A login for ${name} is saved already. Replace it?`, false)))
|
|
805
|
+
return { text: "nothing changed", data: { connector: name, saved: true } };
|
|
806
|
+
if (!(await askLogin(ctx, c)))
|
|
807
|
+
return { text: "no login saved", data: { connector: name, saved: false }, exitCode: 1 };
|
|
808
|
+
return {
|
|
809
|
+
text: lines(`login for ${name} saved — strom puts it into its requests to ${c.manifest.hosts.map(bareHost).join(", ")}`, `forget it: strom login ${name} --remove`),
|
|
810
|
+
data: { connector: name, saved: true },
|
|
811
|
+
};
|
|
812
|
+
},
|
|
813
|
+
});
|
|
814
|
+
function need(v, what) {
|
|
815
|
+
if (!v)
|
|
816
|
+
throw new UsageError(`give ${what}`, { hint: "strom fetch <connector> <book> --images 40-69 --recordset B… · or --find <place>" });
|
|
817
|
+
return v;
|
|
818
|
+
}
|
|
819
|
+
function needImages() {
|
|
820
|
+
throw new UsageError("which images? --images 40-69", { hint: "strom fetch <connector> <book> --list tells how many there are" });
|
|
821
|
+
}
|
|
822
|
+
/** The part of an image --crop and --half name, in fractions of it (pixels: of the registered image). */
|
|
823
|
+
/** A part of one image, of the book the connector fetched this record set from before (or the one named). */
|
|
824
|
+
function partRequest(tree, c, book, opts) {
|
|
825
|
+
if (!opts.recordset)
|
|
826
|
+
throw new UsageError("a part is registered with its image: give --recordset B…", { hint: `strom fetch ${c.name} --recordset B… --images <n> --crop x,y,w,h` });
|
|
827
|
+
const recordset = requireRecord(tree, String(opts.recordset), "recordset").id;
|
|
828
|
+
const images = parseImages(opts.images, 1);
|
|
829
|
+
if (!images)
|
|
830
|
+
throw new UsageError("which image? --images <n> (one)", { hint: `strom fetch ${c.name} --recordset ${recordset} --images <n> --crop x,y,w,h` });
|
|
831
|
+
const all = tree.list("media");
|
|
832
|
+
const whole = findImage(all, recordset, images[0]);
|
|
833
|
+
const known = all.find((m) => m.recordset === recordset && m.fetched?.connector === c.name)?.fetched?.book;
|
|
834
|
+
const b = book ?? known;
|
|
835
|
+
if (!b)
|
|
836
|
+
throw new UsageError(`which book of ${c.name}? none of the images of ${recordset} came through it`, { hint: `strom fetch ${c.name} <book> --recordset ${recordset} --images ${images[0]} --crop x,y,w,h` });
|
|
837
|
+
return { cmd: "part", book: b, image: images[0], region: partRegion(opts, whole?.part ? undefined : whole) };
|
|
838
|
+
}
|
|
839
|
+
/**
|
|
840
|
+
* Images through the user's browser. The connector says where they are (locate,
|
|
841
|
+
* its pages through strom as usual); the limiter reserves a time for each
|
|
842
|
+
* request of the browser — opening the page, then one per image — in the state
|
|
843
|
+
* every strom process shares; the agent runs a script in a tab of the images'
|
|
844
|
+
* site that keeps to those times and saves each image into the downloads folder.
|
|
845
|
+
*/
|
|
846
|
+
/** strom connector test, once the request is known (again after a page from the browser). */
|
|
847
|
+
async function testWith(ctx, c, request, max) {
|
|
848
|
+
if (!(await ensureAllowed(ctx, c)))
|
|
849
|
+
return { text: "not allowed — nothing was sent", exitCode: 1 };
|
|
850
|
+
if (!(await ensureLogin(ctx, c)))
|
|
851
|
+
return { text: "no login — nothing was sent", exitCode: 1 };
|
|
852
|
+
const workDir = path.join(c.dir, ".test");
|
|
853
|
+
fs.mkdirSync(workDir, { recursive: true });
|
|
854
|
+
for (const e of fs.readdirSync(workDir))
|
|
855
|
+
if (e !== "probe")
|
|
856
|
+
fs.rmSync(path.join(workDir, e), { recursive: true, force: true }); // what probes saved stays
|
|
857
|
+
const pages = pagesOf(ctx, c);
|
|
858
|
+
const r = await runConnector(c, request, { env: ctx.env, workDir, netDir: netDir(ctx), maxRequests: max, onLog: (l) => ctx.io.stderr(` · ${l}\n`), ...(pages ? { pages } : {}) });
|
|
859
|
+
if (r.needs)
|
|
860
|
+
return planPages(ctx, c, r.needs, { cmd: "test", request: { ...request }, max }, `strom connector test ${c.name} (${request.cmd})`, describeRun(r, `test ${request.cmd}`));
|
|
861
|
+
const direct = directNetwork(c);
|
|
862
|
+
return {
|
|
863
|
+
text: lines(describeRun(r, `test ${request.cmd}`), r.books.length ? `books (${r.books.length}):\n${bookLines(r, c.name, `strom connector test ${c.name} --list <id>`).join("\n")}` : undefined, r.images.length ? `images (${r.images.length}): ${r.images.map((i) => `${i.n ?? "?"}${i.region ? ` part ${regionText(i.region)}` : ""}=${path.basename(i.file)} ${fs.statSync(i.file).size} B${pixels(i.file)}`).join(", ")} in ${ctx.display(workDir)}` : undefined, r.located.length ? `located (${r.located.length}):\n${r.located.map((l) => ` ${l.n}${l.region ? ` part ${regionText(l.region)}` : ""} ${l.src}${l.url && l.url !== l.src ? `\n page ${l.url}` : ""}`).join("\n")}` : undefined, c.manifest.login ? (loginOf(ctx.env, c) ? "with the user's login" : `without a login${c.manifest.login.required ? "" : " (the user may save one: strom login " + c.name + ")"}`) : undefined, direct.length ? `⚠ the code reaches the network directly — strom cannot pace that:\n${direct.map((d) => ` ${d}`).join("\n")}` : undefined, c.manifest.policy.automation === "unknown" ? "⚠ policy.automation is unknown — find out what the portal's terms say (DISCOVERY.md, step 1)" : undefined),
|
|
864
|
+
data: { ...r, direct },
|
|
865
|
+
...(r.stopped && !r.books.length && !r.images.length && !r.located.length ? { exitCode: 1 } : {}),
|
|
866
|
+
};
|
|
867
|
+
}
|
|
868
|
+
/** What a probe got, saved in .test/probe to read — from strom's request or from the user's browser. */
|
|
869
|
+
function probeResult(ctx, c, method, url, res, save, cookies) {
|
|
870
|
+
const dir = path.join(c.dir, ".test", "probe");
|
|
871
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
872
|
+
const fromUrl = decodeURIComponent((URL.canParse(res.url) ? new URL(res.url) : new URL(url)).pathname.split("/").filter(Boolean).at(-1) ?? "") || "index.html";
|
|
873
|
+
const name = String(save ?? fromUrl).replace(/[^\p{L}\p{M}\p{N}._-]+/gu, "_").replace(/^\.+/, "_");
|
|
874
|
+
const file = path.join(dir, name);
|
|
875
|
+
fs.writeFileSync(file, res.body);
|
|
876
|
+
const textual = /^(text\/|application\/(json|xml|javascript)|image\/svg)/.test(res.contentType);
|
|
877
|
+
const check = botCheck(res.body, res.headers);
|
|
878
|
+
return {
|
|
879
|
+
text: lines(`${method} ${url} → ${res.status} · ${res.contentType || "no type"} · ${res.body.length} B${res.url && res.url !== url ? ` · after redirects ${res.url}` : ""}${cookies ? "" : " · from your browser"}`, ...Object.entries(res.headers)
|
|
880
|
+
.filter(([k]) => !["date", "connection", "keep-alive", "transfer-encoding", "vary"].includes(k))
|
|
881
|
+
.map(([k, v]) => ` ${k}: ${v}`), `saved: ${ctx.display(file)}${textual ? ` — search it: strom connector grep ${c.name} <text> (an address such as /api/, .json, .jpg; a form field)` : ""}`, check
|
|
882
|
+
? c.manifest.browser?.pages
|
|
883
|
+
? `⚠ this is ${check}'s check whether a person is there, not the page: the user passes it in that tab of their browser (never you), then probe again`
|
|
884
|
+
: `⚠ this is ${check}'s check whether a person is there, not the page: the portal gives its pages to a real browser only. Build the connector through the user's browser: in connector.json "routes": ["browser"] and "browser": {"pages": true} (the contract, section 5), then strom connector use ${c.name} --via browser — browser tools come with your next session, and the user passes the check in their own browser, never you`
|
|
885
|
+
: undefined, cookies?.length ? `cookies kept for the next probe: ${cookies.join(", ")} (--fresh starts without them)` : undefined, textual && res.body.length <= 1500 ? res.body.toString("utf8") : undefined),
|
|
886
|
+
data: { status: res.status, type: res.contentType, headers: res.headers, url: res.url, bytes: res.body.length, file, ...(cookies ? { cookies } : { via: "browser" }), ...(check ? { botCheck: check } : {}) },
|
|
887
|
+
};
|
|
888
|
+
}
|
|
889
|
+
/** browser.pages: what the browser got, for the connector's run (undefined: the connector reaches the portal through strom). */
|
|
890
|
+
function pagesOf(ctx, c) {
|
|
891
|
+
if (!c.manifest.browser?.pages)
|
|
892
|
+
return undefined;
|
|
893
|
+
const root = ctx.tree().root;
|
|
894
|
+
return (r) => loadPage(root, c.name, pageKey(r));
|
|
895
|
+
}
|
|
896
|
+
/**
|
|
897
|
+
* A page for a connector whose portal answers a real browser only: planned in
|
|
898
|
+
* strom's limiter, asked by a script in the user's tab, saved into the
|
|
899
|
+
* downloads folder — then --take gives it to the connector and goes on.
|
|
900
|
+
*/
|
|
901
|
+
function planPages(ctx, c, want, resume, what, before) {
|
|
902
|
+
const tree = ctx.tree();
|
|
903
|
+
const u = new URL(want.url);
|
|
904
|
+
const pace = paceOf(c.manifest.policy.pace);
|
|
905
|
+
let times;
|
|
906
|
+
try {
|
|
907
|
+
// one pause ahead: time to open the tab
|
|
908
|
+
times = reserveSlots(netDir(ctx), u.hostname, c.manifest.policy.pace, 1, { leadMs: pace.minIntervalMs });
|
|
909
|
+
}
|
|
910
|
+
catch (err) {
|
|
911
|
+
if (!(err instanceof NetError))
|
|
912
|
+
throw err;
|
|
913
|
+
return { text: lines(before, `nothing planned: ${err.message}`), exitCode: 1 };
|
|
914
|
+
}
|
|
915
|
+
const key = pageKey(want);
|
|
916
|
+
const open = c.manifest.browser?.open && new URL(c.manifest.browser.open).origin === u.origin ? c.manifest.browser.open : `${u.origin}/robots.txt`;
|
|
917
|
+
const page = { key, method: want.method, url: want.url, headers: want.headers, ...(want.body !== undefined ? { body: want.body } : {}), file: `strom-${c.name}-page-${key}`, at: times[0] };
|
|
918
|
+
// a page planned again replaces its older plan
|
|
919
|
+
for (const old of loadPlans(tree.root, c.name))
|
|
920
|
+
if (old.pages?.some((p) => p.key === key))
|
|
921
|
+
removePlan(tree.root, old.id);
|
|
922
|
+
const now = Date.now();
|
|
923
|
+
const book = "request" in resume && typeof resume.request.book === "string" ? resume.request.book : "";
|
|
924
|
+
const plan = { id: `${c.name}-${now}-p`, connector: c.name, book, pages: [page], resume, open, origin: u.origin, host: u.hostname, gap: pace.minIntervalMs, created: now, until: now + PLAN_TTL_MS, items: [] };
|
|
925
|
+
savePlan(tree.root, plan);
|
|
926
|
+
const script = pageScript(plan);
|
|
927
|
+
const take = `strom fetch ${c.name} --take --result "<the line it returned>"`;
|
|
928
|
+
return {
|
|
929
|
+
text: lines(before, `browser plan: 1 page of ${u.hostname} for ${what} — the portal answers a real browser only; about ${Math.max(1, Math.round((page.at - now) / 1000))} s at its pace`, `1. A tab of your own on ${u.origin} — open ${open} if you have none there.`, " The site shows its check whether a person is there (a box to tick, a puzzle), a login, or the browser is not connected: stop and ask the user — that is theirs to do, never yours (strom task wait T… --on \"…\"). Then go on.", "2. Run this in that tab with the JavaScript tool, exactly as it is:", "", script, "", `3. Then: ${take} — strom gives the page to the connector and goes on with ${what}; it may plan the next page: do the same again.`, `The browser saves it into ${ctx.display(ctx.settings.downloads())} as ${page.file}.json — strom takes it over; do not open, move or read it yourself.`),
|
|
930
|
+
data: { plan: { id: plan.id, open, host: plan.host, pages: [{ key, method: page.method, url: page.url, file: page.file, at: new Date(page.at).toISOString() }] }, script, take, resume },
|
|
931
|
+
};
|
|
932
|
+
}
|
|
933
|
+
/** A page the browser got, also as a file of the connector's .test/probe: strom connector grep reads it while the connector is built. */
|
|
934
|
+
function keepForGrep(c, pg, page) {
|
|
935
|
+
const ext = /json/.test(page.contentType) ? ".json" : /javascript/.test(page.contentType) ? ".js" : /^image\/(\w+)/.exec(page.contentType)?.[1] ? `.${/^image\/(\w+)/.exec(page.contentType)[1].replace("jpeg", "jpg")}` : /text\/plain/.test(page.contentType) ? ".txt" : ".html";
|
|
936
|
+
const tail = (URL.canParse(pg.url) ? new URL(pg.url).pathname.split("/").filter(Boolean).at(-1) : undefined)?.replace(/[^\p{L}\p{N}._-]+/gu, "_").slice(0, 60) ?? "page";
|
|
937
|
+
try {
|
|
938
|
+
const dir = path.join(c.dir, ".test", "probe", "browser");
|
|
939
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
940
|
+
fs.writeFileSync(path.join(dir, `${tail}-${pg.key}${ext}`), page.body);
|
|
941
|
+
}
|
|
942
|
+
catch {
|
|
943
|
+
// a copy to read: the page itself is kept in the tree
|
|
944
|
+
}
|
|
945
|
+
}
|
|
946
|
+
/** Pages the browser saved: kept for the connector, and the command they were planned for goes on. */
|
|
947
|
+
async function takePages(ctx, c, plans, result) {
|
|
948
|
+
const tree = ctx.tree();
|
|
949
|
+
const dir = ctx.settings.downloads();
|
|
950
|
+
const got = result === undefined ? [] : parseResult(result);
|
|
951
|
+
const notes = [];
|
|
952
|
+
const taken = [];
|
|
953
|
+
let next;
|
|
954
|
+
for (const plan of plans) {
|
|
955
|
+
const left = [];
|
|
956
|
+
let stop = false;
|
|
957
|
+
for (const pg of plan.pages ?? []) {
|
|
958
|
+
const f = findDownload(dir, pg.file);
|
|
959
|
+
const said = got.find((g) => g.page === pg.key);
|
|
960
|
+
if (!f.file) {
|
|
961
|
+
left.push(pg);
|
|
962
|
+
if (said?.failed)
|
|
963
|
+
notes.push(`${pg.url}: the browser could not ask for it (no network, or the site closed the connection) — run the script again`);
|
|
964
|
+
else if (said?.status === 200)
|
|
965
|
+
notes.push(`${pg.url}: the browser got it but saved nothing into ${ctx.display(dir)} — Chrome lets a site download one file until the user allows automatic downloads for it (at the right end of the address bar)`);
|
|
966
|
+
else if (said?.status && [401, 403, 429].includes(said.status)) {
|
|
967
|
+
const why = refusedBy(netDir(ctx), plan.host, said.status);
|
|
968
|
+
if (why)
|
|
969
|
+
notes.push(why);
|
|
970
|
+
stop = true;
|
|
971
|
+
}
|
|
972
|
+
continue;
|
|
973
|
+
}
|
|
974
|
+
const page = readSavedPage(f.file);
|
|
975
|
+
for (const x of [f.file, ...f.others])
|
|
976
|
+
fs.rmSync(x, { force: true });
|
|
977
|
+
if (typeof page === "string") {
|
|
978
|
+
notes.push(`${pg.url}: ${page} — removed; run the script again`);
|
|
979
|
+
left.push(pg);
|
|
980
|
+
continue;
|
|
981
|
+
}
|
|
982
|
+
const check = botCheck(page.body, page.headers);
|
|
983
|
+
if (check) {
|
|
984
|
+
notes.push(`${plan.host} gave its check whether a person is there (${check}) instead of the page: the user passes it in that tab — never you — then run the same script again`);
|
|
985
|
+
left.push(pg);
|
|
986
|
+
continue;
|
|
987
|
+
}
|
|
988
|
+
if ([401, 403, 429].includes(page.status)) {
|
|
989
|
+
const why = refusedBy(netDir(ctx), plan.host, page.status);
|
|
990
|
+
notes.push(why ?? `${plan.host} answered ${page.status}`);
|
|
991
|
+
stop = true;
|
|
992
|
+
continue;
|
|
993
|
+
}
|
|
994
|
+
savePage(tree.root, c.name, pg, page);
|
|
995
|
+
keepForGrep(c, pg, page);
|
|
996
|
+
taken.push(`${pg.method} ${pg.url} → ${page.status} · ${page.body.length} B`);
|
|
997
|
+
}
|
|
998
|
+
if (stop || !left.length)
|
|
999
|
+
removePlan(tree.root, plan.id);
|
|
1000
|
+
else
|
|
1001
|
+
savePlan(tree.root, { ...plan, pages: left });
|
|
1002
|
+
if (!stop && !left.length && plan.resume)
|
|
1003
|
+
next ??= plan.resume;
|
|
1004
|
+
}
|
|
1005
|
+
const head = lines(taken.length ? `page(s) taken over from your browser: ${taken.join("; ")}` : `no page taken over from ${ctx.display(dir)}`, ...notes.map((n) => ` ⚠ ${n}`));
|
|
1006
|
+
if (!next)
|
|
1007
|
+
return { text: head, data: { taken, notes }, ...(taken.length ? {} : { exitCode: 1 }) };
|
|
1008
|
+
// go on with what the pages were planned for: the connector asks again, the pages it has come from what the browser got
|
|
1009
|
+
let then;
|
|
1010
|
+
if (next.cmd === "probe") {
|
|
1011
|
+
const pg = plans.flatMap((p) => p.pages ?? []).find((p) => loadPage(tree.root, c.name, p.key));
|
|
1012
|
+
const page = pg ? loadPage(tree.root, c.name, pg.key) : undefined;
|
|
1013
|
+
then = page && pg ? probeResult(ctx, c, pg.method, pg.url, page, next.save) : { text: "the probe's page is not there" };
|
|
1014
|
+
}
|
|
1015
|
+
else if (next.cmd === "test")
|
|
1016
|
+
then = await testWith(ctx, c, next.request, next.max);
|
|
1017
|
+
else
|
|
1018
|
+
then = await fetchWith(ctx, c, next.request, next.recordset);
|
|
1019
|
+
return { text: lines(head, then.text), data: { taken, notes, then: then.data }, ...(then.exitCode ? { exitCode: then.exitCode } : {}) };
|
|
1020
|
+
}
|
|
1021
|
+
/** strom fetch, once the request is known: checked against the policy, then through strom or the user's browser. */
|
|
1022
|
+
async function fetchWith(ctx, c, request, recordset) {
|
|
1023
|
+
const tree = ctx.tree();
|
|
1024
|
+
const via = routeOf(ctx.env, c).via;
|
|
1025
|
+
if ((request.cmd === "fetch" || request.cmd === "part") && c.manifest.policy.automation === "manual")
|
|
1026
|
+
throw new UsageError(`${c.manifest.title} does not allow automated download (its terms, as the connector read them)`, {
|
|
1027
|
+
hint: `the user saves them by hand: strom task wait T… --images B…:${request.cmd === "part" ? request.image : runs(request.images)} --on "<the book, its link, which images>" (the link: strom fetch ${c.name} ${request.book} --list)`,
|
|
1028
|
+
});
|
|
1029
|
+
const repo = forbiddenBy(ctx, c);
|
|
1030
|
+
if (repo)
|
|
1031
|
+
throw new UsageError(`${repo.id} ${repo.name} is marked automation ${repo.automation} in this tree — no downloads through a connector`, {
|
|
1032
|
+
hint: `strom repo show ${repo.id} — change it only if the archive allows it: strom repo edit ${repo.id} --automation allowed`,
|
|
1033
|
+
});
|
|
1034
|
+
if (recordset && request.cmd === "fetch") {
|
|
1035
|
+
// what is registered already is not asked for again (a part of an image is not the image)
|
|
1036
|
+
const have = new Set(tree.list("media").filter((m) => m.recordset === recordset && !m.part && m.image !== undefined).map((m) => m.image));
|
|
1037
|
+
const asked = request.images;
|
|
1038
|
+
request.images = asked.filter((n) => !have.has(n));
|
|
1039
|
+
if (!request.images.length)
|
|
1040
|
+
return { text: `images ${runs(asked)} of ${recordset} are registered already — nothing fetched`, data: { added: [], again: asked } };
|
|
1041
|
+
}
|
|
1042
|
+
if (recordset && request.cmd === "part") {
|
|
1043
|
+
const had = tree.list("media").find((m) => m.recordset === recordset && m.image === request.image && m.part && sameRegion(m.part, request.region));
|
|
1044
|
+
if (had)
|
|
1045
|
+
return { text: `part ${regionText(request.region)} of image ${request.image} of ${recordset} is registered already: ${had.id} — nothing fetched`, data: { added: [], again: [had.id] } };
|
|
1046
|
+
}
|
|
1047
|
+
if (tree.dryRun) {
|
|
1048
|
+
// a dry run never contacts the archive, nor asks for consent
|
|
1049
|
+
const what = request.cmd === "find"
|
|
1050
|
+
? `find the books of ${request.place}${request.years ? ` ${request.years}` : ""}`
|
|
1051
|
+
: request.cmd === "list"
|
|
1052
|
+
? `describe book ${request.book}`
|
|
1053
|
+
: request.cmd === "part"
|
|
1054
|
+
? `fetch part ${regionText(request.region)} of image ${request.image} of book ${request.book} as a part of ${recordset}:${request.image}`
|
|
1055
|
+
: `fetch images ${runs(request.images)} of book ${request.book}${recordset ? ` as images of ${recordset}` : " into the inbox"}`;
|
|
1056
|
+
const browser = via === "browser" && (request.cmd === "fetch" || request.cmd === "part");
|
|
1057
|
+
return { text: `dry run: would ${what} through ${c.name}${browser ? ", in your browser" : ""} — nothing was fetched`, data: { dryRun: true, request, recordset, via } };
|
|
1058
|
+
}
|
|
1059
|
+
if (via === "browser" && (request.cmd === "fetch" || request.cmd === "part"))
|
|
1060
|
+
return planBrowser(ctx, c, request, recordset);
|
|
1061
|
+
if (!(await ensureAllowed(ctx, c)))
|
|
1062
|
+
return { text: "not allowed — nothing was fetched", exitCode: 1 };
|
|
1063
|
+
if (!(await ensureLogin(ctx, c)))
|
|
1064
|
+
return { text: "no login — nothing was fetched", exitCode: 1 };
|
|
1065
|
+
if (request.cmd === "fetch")
|
|
1066
|
+
ctx.io.stderr(estimate(c, request.images.length) + "\n");
|
|
1067
|
+
if (request.cmd === "fetch" && c.manifest.policy.automation === "unknown")
|
|
1068
|
+
ctx.io.stderr(`note: what the terms of ${c.manifest.title} say about automated download is not known yet — strom connector show ${c.name}\n`);
|
|
1069
|
+
const workDir = path.join(tree.root, ".strom", "fetch", `${c.name}-${Date.now()}`);
|
|
1070
|
+
const t0 = Date.now();
|
|
1071
|
+
const pages = pagesOf(ctx, c);
|
|
1072
|
+
const r = await runConnector(c, request, { env: ctx.env, workDir, netDir: netDir(ctx), onLog: (l) => ctx.io.stderr(` · ${l}\n`), ...(pages ? { pages } : {}) });
|
|
1073
|
+
if (r.needs) {
|
|
1074
|
+
fs.rmSync(workDir, { recursive: true, force: true });
|
|
1075
|
+
return planPages(ctx, c, r.needs, { cmd: "fetch", request: { ...request }, ...(recordset ? { recordset } : {}) }, `strom fetch ${c.name} (${request.cmd})`, describeRun(r, request.cmd));
|
|
1076
|
+
}
|
|
1077
|
+
const took = `${Math.round((Date.now() - t0) / 1000)} s`;
|
|
1078
|
+
if (request.cmd === "find" || request.cmd === "list") {
|
|
1079
|
+
fs.rmSync(workDir, { recursive: true, force: true });
|
|
1080
|
+
return {
|
|
1081
|
+
text: lines(describeRun(r, `${request.cmd} (${took})`), r.books.length ? [`books (${r.books.length}):`, ...bookLines(r, c.name, `strom fetch ${c.name} <id> --list`)].join("\n") : "no books found"),
|
|
1082
|
+
data: r,
|
|
1083
|
+
...(r.stopped && !r.books.length ? { exitCode: 1 } : {}),
|
|
1084
|
+
};
|
|
1085
|
+
}
|
|
1086
|
+
const from = (f) => `connector ${c.name} · ${path.basename(f)}`;
|
|
1087
|
+
const fetched = { connector: c.name, book: request.book };
|
|
1088
|
+
let text;
|
|
1089
|
+
let data = { ...r };
|
|
1090
|
+
if (request.cmd === "part") {
|
|
1091
|
+
const res = registerImages(tree, shared(ctx), r.images.map((i) => ({ file: i.file, image: i.n, url: i.url, from: from(i.file), part: i.region, fetched })), recordset);
|
|
1092
|
+
fs.rmSync(workDir, { recursive: true, force: true });
|
|
1093
|
+
const m = res.added[0];
|
|
1094
|
+
const whole = recordset ? findImage(tree.list("media").filter((x) => !x.part), recordset, request.image) : undefined;
|
|
1095
|
+
const gain = m?.width && m.part && whole?.width ? m.width / m.part.w / whole.width : undefined;
|
|
1096
|
+
text = m
|
|
1097
|
+
? lines(`part ${regionText(m.part)} of image ${request.image} of ${recordset} fetched and registered: ${m.id}${m.width ? ` · ${m.width}×${m.height} px` : ""}${gain ? ` · ${gain.toFixed(1)}× the detail of the whole image` : ""} · ${took}`, gain !== undefined && gain < 1.2 ? " no sharper than the whole image: the portal gives no more detail than that" : undefined, `look at it: strom media view ${recordset}:${request.image} --crop ${regionText(m.part)} (a view of the image uses the part by itself)`)
|
|
1098
|
+
: res.again.length
|
|
1099
|
+
? lines(`that part is registered already: ${res.again.join(" ")}`, ...res.clashes.map((x) => `⚠ ${x}`))
|
|
1100
|
+
: "no part fetched";
|
|
1101
|
+
data = { ...data, added: res.added.map((x) => ({ id: x.id, image: x.image, part: x.part })), again: res.again };
|
|
1102
|
+
}
|
|
1103
|
+
else if (recordset && r.images.length) {
|
|
1104
|
+
const res = registerImages(tree, shared(ctx), r.images.map((i) => ({ file: i.file, image: i.n, url: i.url, from: from(i.file), fetched })), recordset);
|
|
1105
|
+
const nums = res.added.map((m) => m.image).filter((n) => n !== undefined);
|
|
1106
|
+
text = lines(`${res.added.length} image(s) of ${recordset}${nums.length ? ` (images ${Math.min(...nums)}–${Math.max(...nums)})` : ""} fetched and registered · ${took}`, res.again.length ? `${res.again.length} already registered` : undefined, ...res.clashes.slice(0, 10).map((x) => `⚠ ${x}`), res.woken.length ? `back in the queue (they waited for these images): ${res.woken.join(" ")}` : undefined);
|
|
1107
|
+
data = { ...data, added: res.added.map((m) => ({ id: m.id, image: m.image })), again: res.again, woken: res.woken, clashes: res.clashes };
|
|
1108
|
+
fs.rmSync(workDir, { recursive: true, force: true });
|
|
1109
|
+
}
|
|
1110
|
+
else if (r.images.length) {
|
|
1111
|
+
// into the inbox, one folder for the book: the usual way from there
|
|
1112
|
+
const folder = `${c.name} ${String(request.book).replace(/[^\p{L}\p{M}\p{N}._-]+/gu, "_")}`;
|
|
1113
|
+
const inbox = path.join(shared(ctx), "inbox", folder);
|
|
1114
|
+
fs.mkdirSync(inbox, { recursive: true });
|
|
1115
|
+
for (const i of r.images)
|
|
1116
|
+
fs.renameSync(i.file, path.join(inbox, path.basename(i.file)));
|
|
1117
|
+
fs.rmSync(workDir, { recursive: true, force: true });
|
|
1118
|
+
text = lines(`${r.images.length} image(s) fetched into the inbox: ${folder}/ · ${took}`, `register them: strom media add --inbox ${shellArg(folder)} --recordset B…`);
|
|
1119
|
+
data = { ...data, inbox: folder };
|
|
1120
|
+
}
|
|
1121
|
+
else {
|
|
1122
|
+
fs.rmSync(workDir, { recursive: true, force: true });
|
|
1123
|
+
text = "no images fetched";
|
|
1124
|
+
}
|
|
1125
|
+
return { text: lines(describeRun(r, request.cmd), text), data, ...(r.stopped && !r.images.length ? { exitCode: 1 } : {}) };
|
|
1126
|
+
}
|
|
1127
|
+
async function planBrowser(ctx, c, request, recordset) {
|
|
1128
|
+
const tree = ctx.tree();
|
|
1129
|
+
if (!c.manifest.can.includes("locate"))
|
|
1130
|
+
throw new UsageError(`connector ${c.name} cannot say where its images are (can: locate) — the browser has no addresses to fetch`, { hint: `strom connector use ${c.name} --via direct` });
|
|
1131
|
+
if (!(await ensureAllowed(ctx, c)))
|
|
1132
|
+
return { text: "not allowed — nothing was planned", exitCode: 1 };
|
|
1133
|
+
const pace = paceOf(c.manifest.policy.pace);
|
|
1134
|
+
const part = request.cmd === "part";
|
|
1135
|
+
const asked = part ? [request.image] : request.images;
|
|
1136
|
+
// one script waits at most SCRIPT_MS: as many images as fit in it, the rest next time
|
|
1137
|
+
const batch = asked.slice(0, Math.max(1, Math.floor(SCRIPT_MS / pace.minIntervalMs)));
|
|
1138
|
+
const workDir = path.join(tree.root, ".strom", "fetch", `${c.name}-${Date.now()}`);
|
|
1139
|
+
const pages = pagesOf(ctx, c);
|
|
1140
|
+
const r = await runConnector(c, { cmd: "locate", book: request.book, images: batch, ...(part ? { region: request.region } : {}) }, { env: ctx.env, workDir, netDir: netDir(ctx), onLog: (l) => ctx.io.stderr(` · ${l}\n`), ...(pages ? { pages } : {}) });
|
|
1141
|
+
fs.rmSync(workDir, { recursive: true, force: true });
|
|
1142
|
+
if (r.needs)
|
|
1143
|
+
return planPages(ctx, c, r.needs, { cmd: "fetch", request: { ...request }, ...(recordset ? { recordset } : {}) }, `strom fetch ${c.name} (${request.cmd} of book ${request.book})`, describeRun(r, "locate"));
|
|
1144
|
+
if (!r.located.length)
|
|
1145
|
+
return { text: lines(describeRun(r, "locate"), "no image located — nothing planned"), data: r, exitCode: 1 };
|
|
1146
|
+
// one site at a time: the script fetches from the page it runs on
|
|
1147
|
+
const first = new URL(r.located[0].src);
|
|
1148
|
+
const here = batch.map((n) => r.located.find((l) => l.n === n && new URL(l.src).origin === first.origin)).filter((l) => l !== undefined);
|
|
1149
|
+
const open = c.manifest.browser?.open && new URL(c.manifest.browser.open).origin === first.origin ? c.manifest.browser.open : `${first.origin}/robots.txt`;
|
|
1150
|
+
let times;
|
|
1151
|
+
try {
|
|
1152
|
+
times = reserveSlots(netDir(ctx), first.hostname, c.manifest.policy.pace, here.length + 1);
|
|
1153
|
+
}
|
|
1154
|
+
catch (err) {
|
|
1155
|
+
if (!(err instanceof NetError))
|
|
1156
|
+
throw err;
|
|
1157
|
+
return { text: lines(describeRun(r, "locate"), `nothing planned: ${err.message}`), data: r, exitCode: 1 };
|
|
1158
|
+
}
|
|
1159
|
+
const items = here
|
|
1160
|
+
.slice(0, times.length - 1)
|
|
1161
|
+
.map((l, i) => ({ n: l.n, src: l.src, url: l.url ?? l.src, ...(l.page ? { page: l.page } : {}), ...(l.region ? { region: l.region } : {}), file: fileBase(c.name, request.book, l.n, part), at: times[i + 1] }))
|
|
1162
|
+
.filter((it, i) => i === 0 || it.at - times[1] <= SCRIPT_MS);
|
|
1163
|
+
const now = Date.now();
|
|
1164
|
+
const plan = { id: `${c.name}-${now}`, connector: c.name, book: request.book, ...(recordset ? { recordset } : {}), open, origin: first.origin, host: first.hostname, gap: times.length > 1 ? times[1] - times[0] : pace.minIntervalMs, created: now, until: now + PLAN_TTL_MS, items };
|
|
1165
|
+
// an image planned again replaces its older plan
|
|
1166
|
+
for (const old of loadPlans(tree.root, c.name)) {
|
|
1167
|
+
const left = old.items.filter((i) => !(old.book === plan.book && items.some((x) => x.file === i.file)));
|
|
1168
|
+
if (left.length === old.items.length)
|
|
1169
|
+
continue;
|
|
1170
|
+
if (left.length)
|
|
1171
|
+
savePlan(tree.root, { ...old, items: left });
|
|
1172
|
+
else
|
|
1173
|
+
removePlan(tree.root, old.id);
|
|
1174
|
+
}
|
|
1175
|
+
savePlan(tree.root, plan);
|
|
1176
|
+
const script = planScript(plan);
|
|
1177
|
+
const rest = asked.filter((n) => !items.some((i) => i.n === n));
|
|
1178
|
+
const secs = Math.max(1, Math.round((items.at(-1).at - now) / 1000));
|
|
1179
|
+
const take = `strom fetch ${c.name} --take --result "<the line it returned>"`;
|
|
1180
|
+
const shown = `${items[0].file}.jpg${items.length > 1 ? " …" : ""}`;
|
|
1181
|
+
return {
|
|
1182
|
+
text: lines(describeRun(r, "locate"), `browser plan: ${part ? `part ${regionText(request.region)} of image ${request.image}` : `${items.length} image(s) (${runs(items.map((i) => i.n))})`} of book ${request.book} through your browser — about ${secs} s at the pace of ${first.hostname}`, `1. Open a new tab of your own at ${open}${open.endsWith("/robots.txt") ? " — a light page of the site: the script fetches the images from there" : ""}.`, ' The browser not connected, a login or a captcha on the way: stop and ask the user (strom task wait T… --on "…") — that is theirs to do.', " (Not connected: Chrome must be running with the Claude extension; after a change of network the extension may need a click on its icon, or Chrome a restart.)", "2. Run this in that tab with the JavaScript tool, exactly as it is — it waits between the images by itself:", "", script, "", `3. Then: ${take}`, `The browser saves them into ${ctx.display(ctx.settings.downloads())} as ${shown} — strom takes them over; do not open, move or read them yourself.`, `The first time, Chrome asks whether ${first.hostname} may download several files: the user allows it once (at the right end of the address bar).`, rest.length ? `The rest (images ${runs(rest)}): the same strom fetch again, after the take.` : undefined),
|
|
1183
|
+
data: { ...r, plan: { id: plan.id, open, host: plan.host, items: items.map((i) => ({ n: i.n, src: i.src, file: i.file, at: new Date(i.at).toISOString() })) }, script, take, rest },
|
|
1184
|
+
};
|
|
1185
|
+
}
|
|
1186
|
+
/** What the browser downloaded: taken over from the downloads folder, checked, registered (or put into the inbox). */
|
|
1187
|
+
async function takeOver(ctx, c, result) {
|
|
1188
|
+
const tree = ctx.tree();
|
|
1189
|
+
const all = loadPlans(tree.root, c.name);
|
|
1190
|
+
// pages for the connector first: what they were planned for goes on
|
|
1191
|
+
const pagePlans = all.filter((p) => p.pages?.length);
|
|
1192
|
+
if (pagePlans.length && !tree.dryRun)
|
|
1193
|
+
return takePages(ctx, c, pagePlans, result);
|
|
1194
|
+
const plans = all.filter((p) => p.items.length);
|
|
1195
|
+
if (!plans.length)
|
|
1196
|
+
return { text: `nothing of ${c.name} waits to be taken over — plan it first: strom fetch ${c.name} <book> --images <from-to> --recordset B…`, data: { taken: [] } };
|
|
1197
|
+
const dir = ctx.settings.downloads();
|
|
1198
|
+
const got = parseResult(result ?? "");
|
|
1199
|
+
if (tree.dryRun) {
|
|
1200
|
+
const found = plans.flatMap((p) => p.items.filter((i) => findDownload(dir, i.file).file));
|
|
1201
|
+
return { text: `dry run: would take over ${found.length} of ${plans.reduce((n, p) => n + p.items.length, 0)} planned image(s) from ${ctx.display(dir)} — nothing was changed`, data: { dryRun: true, found: found.map((i) => i.n) } };
|
|
1202
|
+
}
|
|
1203
|
+
const notes = [];
|
|
1204
|
+
// what the archive said to the browser counts as if it had said it to strom
|
|
1205
|
+
for (const g of got) {
|
|
1206
|
+
const plan = plans.find((p) => p.items.some((i) => i.n === g.n)) ?? plans.at(-1);
|
|
1207
|
+
if (g.failed)
|
|
1208
|
+
notes.push(`image ${g.n}: the browser could not fetch it — no answer, or the site does not let this page fetch it`);
|
|
1209
|
+
else if (g.status && g.status >= 400)
|
|
1210
|
+
notes.push(refusedBy(netDir(ctx), plan.host, g.status) ?? `image ${g.n}: HTTP ${g.status}`);
|
|
1211
|
+
}
|
|
1212
|
+
const taken = [];
|
|
1213
|
+
const missing = [];
|
|
1214
|
+
for (const plan of plans)
|
|
1215
|
+
for (const item of plan.items) {
|
|
1216
|
+
const f = findDownload(dir, item.file);
|
|
1217
|
+
if (!f.file) {
|
|
1218
|
+
missing.push({ plan, item, unfinished: !!f.unfinished });
|
|
1219
|
+
continue;
|
|
1220
|
+
}
|
|
1221
|
+
const problem = imageProblem(f.file, item.region ? "part" : "whole");
|
|
1222
|
+
if (problem) {
|
|
1223
|
+
// a broken download of strom's own name: gone, so that the next one gets the name
|
|
1224
|
+
for (const x of [f.file, ...f.others])
|
|
1225
|
+
fs.rmSync(x, { force: true });
|
|
1226
|
+
notes.push(`image ${item.n} (${path.basename(f.file)}): ${problem} — removed; plan it again`);
|
|
1227
|
+
missing.push({ plan, item, unfinished: false });
|
|
1228
|
+
continue;
|
|
1229
|
+
}
|
|
1230
|
+
taken.push({ plan, item, file: f.file, others: f.others });
|
|
1231
|
+
}
|
|
1232
|
+
const added = [];
|
|
1233
|
+
const again = [];
|
|
1234
|
+
const woken = [];
|
|
1235
|
+
const inboxed = [];
|
|
1236
|
+
for (const plan of plans) {
|
|
1237
|
+
const mine = taken.filter((t) => t.plan === plan);
|
|
1238
|
+
if (!mine.length)
|
|
1239
|
+
continue;
|
|
1240
|
+
const fetched = { connector: c.name, book: plan.book, via: "browser" };
|
|
1241
|
+
if (plan.recordset) {
|
|
1242
|
+
const res = registerImages(tree, shared(ctx), mine.map((t) => ({ file: t.file, image: t.item.n, url: t.item.url, from: `connector ${c.name} · your browser · ${path.basename(t.file)}`, part: t.item.region, fetched })), plan.recordset);
|
|
1243
|
+
added.push(...res.added.map((m) => ({ id: m.id, ...(m.image !== undefined ? { image: m.image } : {}) })));
|
|
1244
|
+
again.push(...res.again);
|
|
1245
|
+
notes.push(...res.clashes);
|
|
1246
|
+
woken.push(...res.woken);
|
|
1247
|
+
for (const t of mine)
|
|
1248
|
+
for (const x of [t.file, ...t.others])
|
|
1249
|
+
fs.rmSync(x, { force: true });
|
|
1250
|
+
}
|
|
1251
|
+
else {
|
|
1252
|
+
// into the inbox, one folder for the book, as a direct fetch does
|
|
1253
|
+
const folder = `${c.name} ${String(plan.book).replace(/[^\p{L}\p{M}\p{N}._-]+/gu, "_")}`;
|
|
1254
|
+
const inbox = path.join(shared(ctx), "inbox", folder);
|
|
1255
|
+
fs.mkdirSync(inbox, { recursive: true });
|
|
1256
|
+
for (const t of mine) {
|
|
1257
|
+
const to = path.join(inbox, `s${String(t.item.n).padStart(4, "0")}${path.extname(t.file).toLowerCase()}`);
|
|
1258
|
+
fs.copyFileSync(t.file, to);
|
|
1259
|
+
for (const x of [t.file, ...t.others])
|
|
1260
|
+
fs.rmSync(x, { force: true });
|
|
1261
|
+
}
|
|
1262
|
+
inboxed.push(folder);
|
|
1263
|
+
}
|
|
1264
|
+
const left = plan.items.filter((i) => !mine.some((t) => t.item === i));
|
|
1265
|
+
if (left.length)
|
|
1266
|
+
savePlan(tree.root, { ...plan, items: left });
|
|
1267
|
+
else
|
|
1268
|
+
removePlan(tree.root, plan.id);
|
|
1269
|
+
}
|
|
1270
|
+
const fetchedOk = new Set(got.filter((g) => g.status === 200).map((g) => g.n));
|
|
1271
|
+
const lost = missing.filter((m) => fetchedOk.has(m.item.n) && !m.unfinished);
|
|
1272
|
+
const notYet = missing.filter((m) => !fetchedOk.has(m.item.n) || m.unfinished);
|
|
1273
|
+
const nums = (xs) => runs(xs.map((x) => x.item.n));
|
|
1274
|
+
return {
|
|
1275
|
+
text: lines(taken.length
|
|
1276
|
+
? `${taken.length} image(s) taken over from ${ctx.display(dir)}${added.length ? `: registered ${added.map((a) => a.id).join(" ")}` : ""}${inboxed.length ? ` into the inbox: ${[...new Set(inboxed)].join(", ")}/ — register them: strom media add --inbox ${shellArg(inboxed[0])} --recordset B…` : ""}`
|
|
1277
|
+
: `nothing taken over from ${ctx.display(dir)}`, again.length ? ` already registered: ${again.join(" ")}` : undefined, woken.length ? ` back in the queue (they waited for these images): ${woken.join(" ")}` : undefined, ...notes.map((n) => ` ⚠ ${n}`), lost.length
|
|
1278
|
+
? ` images ${nums(lost)}: the browser fetched them but saved nothing here — Chrome lets a site download one file until the user allows automatic downloads for it (at the right end of the address bar), or it asks where to save each file (Settings → Downloads). Then plan them again: strom fetch ${c.name} ${lost[0].plan.book} --images ${nums(lost)}${lost[0].plan.recordset ? ` --recordset ${lost[0].plan.recordset}` : ""}`
|
|
1279
|
+
: undefined, notYet.length ? ` still waiting for images ${nums(notYet)}${notYet.some((m) => m.unfinished) ? " (a download is still running)" : ""} — run the script of their plan, or they went elsewhere: strom config set browser.downloads <folder>` : undefined),
|
|
1280
|
+
data: { taken: taken.map((t) => ({ n: t.item.n, file: t.file })), added, again, woken, inbox: [...new Set(inboxed)], missing: missing.map((m) => m.item.n), notes },
|
|
1281
|
+
...(!taken.length && missing.length ? { exitCode: 1 } : {}),
|
|
1282
|
+
};
|
|
1283
|
+
}
|
|
1284
|
+
/** The fields of a saved login, in the order the connector asks for them. */
|
|
1285
|
+
function fieldsOf(values, c) {
|
|
1286
|
+
const order = Object.keys(c?.manifest.login?.fields ?? {});
|
|
1287
|
+
return Object.keys(values).sort((a, b) => (order.indexOf(a) + 1 || 99) - (order.indexOf(b) + 1 || 99));
|
|
1288
|
+
}
|
|
1289
|
+
function loginState(ctx, c) {
|
|
1290
|
+
const l = loginOf(ctx.env, c);
|
|
1291
|
+
if (l)
|
|
1292
|
+
return `saved ${l.at.slice(0, 10)} (${fieldsOf(l.values, c).join(", ")})`;
|
|
1293
|
+
return c.manifest.login?.required ? `needed, none saved — the user saves it: strom login ${c.name}` : `none — the user may save one: strom login ${c.name}`;
|
|
1294
|
+
}
|
|
1295
|
+
/** A connector that cannot work without the user's login: the user, on a terminal, types it in right here. */
|
|
1296
|
+
async function ensureLogin(ctx, c) {
|
|
1297
|
+
if (!c.manifest.login?.required || loginOf(ctx.env, c))
|
|
1298
|
+
return true;
|
|
1299
|
+
ctx.requireHuman(`Connector ${c.name} needs your login to ${c.manifest.title} (${c.manifest.login.about}) — save it?`, `strom login ${c.name}`, `login:${c.name}`, undefined, { window: false });
|
|
1300
|
+
return askLogin(ctx, c);
|
|
1301
|
+
}
|
|
1302
|
+
/** The user types in their login: shown what it is for and where it goes first; hidden fields are not shown as typed. */
|
|
1303
|
+
async function askLogin(ctx, c) {
|
|
1304
|
+
const l = c.manifest.login;
|
|
1305
|
+
ctx.io.stdout(lines("", `Your login to ${c.manifest.title}, for connector ${c.name}`, ` · what it gives: ${l.about}${l.url ? `\n · an account: ${l.url}` : ""}`, ` · kept on this computer only, readable by you alone (${ctx.display(loginsFile(ctx.env))}) — never in a tree, the shared folder or git`, ` · strom puts it into the connector's requests itself, only to ${c.manifest.hosts.map(bareHost).join(", ")} and only over https;`, " the connector's code never sees it, nor does your agent", "") + "\n");
|
|
1306
|
+
const values = {};
|
|
1307
|
+
for (const [field, label] of Object.entries(l.fields)) {
|
|
1308
|
+
const v = VISIBLE_FIELDS.has(field) ? await ctx.ask(`${label}:`) : await ctx.askSecret(`${label} (not shown):`);
|
|
1309
|
+
if (!v) {
|
|
1310
|
+
ctx.io.stdout("nothing typed — nothing saved\n");
|
|
1311
|
+
return false;
|
|
1312
|
+
}
|
|
1313
|
+
values[field] = v; // as typed: a password is compared byte for byte
|
|
1314
|
+
}
|
|
1315
|
+
saveLogin(ctx.env, c.name, { dir: c.dir, hosts: c.manifest.hosts.map(bareHost), values, at: new Date().toISOString() });
|
|
1316
|
+
return true;
|
|
1317
|
+
}
|
|
1318
|
+
function requestOf(opts, test) {
|
|
1319
|
+
if (opts.find)
|
|
1320
|
+
return { cmd: "find", place: String(opts.find).normalize("NFC"), ...(opts.years ? { years: String(opts.years) } : {}) };
|
|
1321
|
+
if (test && opts.list)
|
|
1322
|
+
return { cmd: "list", book: String(opts.list) };
|
|
1323
|
+
if (test && opts.fetch && (opts.crop !== undefined || opts.half !== undefined)) {
|
|
1324
|
+
const images = parseImages(opts.images, 1) ?? [1];
|
|
1325
|
+
return { cmd: "part", book: String(opts.fetch), image: images[0], region: partRegion(opts) };
|
|
1326
|
+
}
|
|
1327
|
+
if (test && opts.fetch)
|
|
1328
|
+
return { cmd: "fetch", book: String(opts.fetch), images: parseImages(opts.images, 10) ?? [1] };
|
|
1329
|
+
if (test && opts.locate) {
|
|
1330
|
+
const part = opts.crop !== undefined || opts.half !== undefined;
|
|
1331
|
+
return { cmd: "locate", book: String(opts.locate), images: parseImages(opts.images, part ? 1 : 10) ?? [1], ...(part ? { region: partRegion(opts) } : {}) };
|
|
1332
|
+
}
|
|
1333
|
+
throw new UsageError(test ? "what to try: --find <place>, --list <book>, --fetch <book> --images 1-2 or --locate <book> --images 1-2" : "give a book, or --find <place>");
|
|
1334
|
+
}
|
|
1335
|
+
/** A search pattern, found in any case: the text as it is, or a regular expression. */
|
|
1336
|
+
function grepPattern(p, regex) {
|
|
1337
|
+
// as typed, composed or not; a space is any white space (a page's "Kniha 26" reads with one)
|
|
1338
|
+
const text = (t) => t.trim().split(/\s+/u).map((w) => w.replace(/[.*+?^${}()|[\]\\/]/g, "\\$&")).join("\\s+");
|
|
1339
|
+
if (!regex)
|
|
1340
|
+
return new RegExp([...new Set([p.normalize("NFC"), p.normalize("NFD")])].map(text).join("|"), "giu");
|
|
1341
|
+
try {
|
|
1342
|
+
return new RegExp(p, "gi");
|
|
1343
|
+
}
|
|
1344
|
+
catch (e) {
|
|
1345
|
+
throw new UsageError(`not a regular expression: ${p} — ${e.message}`);
|
|
1346
|
+
}
|
|
1347
|
+
}
|
|
1348
|
+
/** The saved files of a connector's .test folder that hold text (not images, not strom's own dotfiles). */
|
|
1349
|
+
function textFiles(dir) {
|
|
1350
|
+
const out = [];
|
|
1351
|
+
const walk = (d) => {
|
|
1352
|
+
let entries;
|
|
1353
|
+
try {
|
|
1354
|
+
entries = fs.readdirSync(d, { withFileTypes: true });
|
|
1355
|
+
}
|
|
1356
|
+
catch {
|
|
1357
|
+
return;
|
|
1358
|
+
}
|
|
1359
|
+
for (const e of entries.sort((a, b) => a.name.localeCompare(b.name))) {
|
|
1360
|
+
if (e.name.startsWith("."))
|
|
1361
|
+
continue;
|
|
1362
|
+
const p = path.join(d, e.name);
|
|
1363
|
+
if (e.isDirectory())
|
|
1364
|
+
walk(p);
|
|
1365
|
+
else if (e.isFile() && fs.statSync(p).size <= 50 * 1024 * 1024) {
|
|
1366
|
+
const fd = fs.openSync(p, "r");
|
|
1367
|
+
const head = Buffer.alloc(4096);
|
|
1368
|
+
const n = fs.readSync(fd, head, 0, head.length, 0);
|
|
1369
|
+
fs.closeSync(fd);
|
|
1370
|
+
if (!head.subarray(0, n).includes(0))
|
|
1371
|
+
out.push(p);
|
|
1372
|
+
}
|
|
1373
|
+
}
|
|
1374
|
+
};
|
|
1375
|
+
walk(dir);
|
|
1376
|
+
return out;
|
|
1377
|
+
}
|