strom-research 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +373 -0
- package/README.md +142 -0
- package/assets/lang/cs.json +302 -0
- package/assets/lang/de.json +302 -0
- package/assets/method/core.md +43 -0
- package/assets/method/enrich.md +11 -0
- package/assets/method/intake.md +30 -0
- package/assets/method/link.md +28 -0
- package/assets/method/locate.md +28 -0
- package/assets/method/narrate.md +13 -0
- package/assets/method/reading.md +62 -0
- package/assets/method/recording.md +59 -0
- package/assets/method/request.md +10 -0
- package/assets/method/verify.md +17 -0
- package/assets/plugins/README.md +23 -0
- package/assets/plugins/connectors/DISCOVERY.md +159 -0
- package/assets/plugins/connectors/README.md +376 -0
- package/assets/plugins/connectors/sdk.ts +168 -0
- package/assets/plugins/connectors/template.ts +38 -0
- package/assets/plugins/gitignore +4 -0
- package/dist/agents/files.js +313 -0
- package/dist/agents/global.js +257 -0
- package/dist/agents/launch.js +36 -0
- package/dist/agents/profiles.js +95 -0
- package/dist/brief/brief.js +345 -0
- package/dist/cli/commit.js +44 -0
- package/dist/cli/context.js +311 -0
- package/dist/cli/execute.js +154 -0
- package/dist/cli/fixes.js +78 -0
- package/dist/cli/format.js +53 -0
- package/dist/cli/help.js +59 -0
- package/dist/cli/main.js +152 -0
- package/dist/cli/menu.js +212 -0
- package/dist/cli/registry.js +96 -0
- package/dist/cli/ui.js +266 -0
- package/dist/cli/wizard.js +142 -0
- package/dist/cli.js +14 -0
- package/dist/commands/analysis.js +622 -0
- package/dist/commands/batch.js +181 -0
- package/dist/commands/checks.js +153 -0
- package/dist/commands/connectors.js +1377 -0
- package/dist/commands/guide.js +160 -0
- package/dist/commands/index.js +19 -0
- package/dist/commands/intake.js +234 -0
- package/dist/commands/media.js +406 -0
- package/dist/commands/meta.js +195 -0
- package/dist/commands/output.js +117 -0
- package/dist/commands/people.js +664 -0
- package/dist/commands/read.js +199 -0
- package/dist/commands/research.js +139 -0
- package/dist/commands/session.js +605 -0
- package/dist/commands/setup.js +465 -0
- package/dist/commands/sources.js +634 -0
- package/dist/commands/start.js +383 -0
- package/dist/commands/story.js +75 -0
- package/dist/commands/tasks.js +436 -0
- package/dist/commands/trees.js +128 -0
- package/dist/core/actions.js +852 -0
- package/dist/core/age.js +95 -0
- package/dist/core/apps.js +74 -0
- package/dist/core/assets.js +34 -0
- package/dist/core/awake.js +33 -0
- package/dist/core/browser.js +281 -0
- package/dist/core/calibration.js +48 -0
- package/dist/core/check.js +112 -0
- package/dist/core/chromium.js +88 -0
- package/dist/core/config.js +348 -0
- package/dist/core/connector.js +811 -0
- package/dist/core/deps.js +73 -0
- package/dist/core/dialog.js +61 -0
- package/dist/core/errors.js +89 -0
- package/dist/core/evidence.js +58 -0
- package/dist/core/frontier.js +219 -0
- package/dist/core/gdate.js +77 -0
- package/dist/core/git.js +300 -0
- package/dist/core/guard.js +124 -0
- package/dist/core/http2.js +76 -0
- package/dist/core/import.js +541 -0
- package/dist/core/install.js +28 -0
- package/dist/core/integrity.js +219 -0
- package/dist/core/json.js +87 -0
- package/dist/core/lang.js +70 -0
- package/dist/core/live.js +244 -0
- package/dist/core/lock.js +112 -0
- package/dist/core/logins.js +67 -0
- package/dist/core/media.js +223 -0
- package/dist/core/model.js +101 -0
- package/dist/core/net.js +366 -0
- package/dist/core/open.js +29 -0
- package/dist/core/paths.js +84 -0
- package/dist/core/people.js +283 -0
- package/dist/core/phrases.js +85 -0
- package/dist/core/queue.js +113 -0
- package/dist/core/reader.js +76 -0
- package/dist/core/records.js +105 -0
- package/dist/core/roles.js +30 -0
- package/dist/core/schema.js +261 -0
- package/dist/core/seal.js +77 -0
- package/dist/core/self.js +40 -0
- package/dist/core/session.js +155 -0
- package/dist/core/shortcut.js +90 -0
- package/dist/core/stories.js +61 -0
- package/dist/core/stromapp.js +138 -0
- package/dist/core/text.js +104 -0
- package/dist/core/tree.js +507 -0
- package/dist/core/uninstall.js +128 -0
- package/dist/core/update.js +193 -0
- package/dist/core/validate.js +260 -0
- package/dist/core/views.js +164 -0
- package/dist/core/which.js +51 -0
- package/dist/core/workers.js +42 -0
- package/dist/gedcom/export.js +454 -0
- package/dist/gedcom/labels.js +103 -0
- package/dist/gedcom/lines.js +91 -0
- package/dist/gedcom/parse.js +53 -0
- package/dist/gedcom/validate.js +183 -0
- package/dist/image/image.js +223 -0
- package/dist/image/index.js +114 -0
- package/dist/image/jpeg-decode.js +552 -0
- package/dist/image/jpeg-encode.js +254 -0
- package/dist/image/png.js +241 -0
- package/dist/runners/antigravity.js +70 -0
- package/dist/runners/claude.js +179 -0
- package/dist/runners/codex.js +45 -0
- package/dist/runners/index.js +13 -0
- package/dist/runners/jsonl.js +86 -0
- package/dist/runners/opencode.js +50 -0
- package/dist/runners/runner.js +63 -0
- package/dist/runners/script.js +58 -0
- package/package.json +44 -0
package/dist/core/age.js
ADDED
|
@@ -0,0 +1,95 @@
|
|
|
1
|
+
// Ages as records state them ("aged 27", "27 let", "3 Monate") in the GEDCOM
|
|
2
|
+
// 5.5.1 form: "27y", "27y 3m", "3m 12d", "<1y", ">60y", INFANT, STILLBORN,
|
|
3
|
+
// CHILD. An age at an event is the most common clue to a birth year.
|
|
4
|
+
const KEYWORDS = new Set(["INFANT", "STILLBORN", "CHILD"]);
|
|
5
|
+
/** Words for years, months, weeks and days in the languages of old records and of users. */
|
|
6
|
+
const UNITS = [
|
|
7
|
+
[/^(y|yr|yrs|year|years|r|rok|roku|roky|let|l|j|jahr|jahre|jahren|lat|lata|rok[uó]w|an|ans|anni|anno|annorum|annos|años|ano|anos|år|ar|év|ev|ann)\.?$/i, "y"],
|
|
8
|
+
[/^(m|mo|mos|month|months|měs|mes|měsíc|měsíce|měsíců|mesic|mesice|mesicu|monat|monate|monaten|mies|miesiąc|miesiące|miesięcy|mois|mese|mesi|meses|mån|hónap)\.?$/i, "m"],
|
|
9
|
+
[/^(w|wk|wks|week|weeks|t|týd|týden|týdny|týdnů|tyden|tydny|woche|wochen|tydz|tydzień|tygodnie|tygodni|sem|semaine|semaines|settimana|settimane)\.?$/i, "w"],
|
|
10
|
+
[/^(d|day|days|den|dny|dní|dni|dnů|tag|tage|tagen|dzień|jour|jours|giorno|giorni|días|dias)\.?$/i, "d"],
|
|
11
|
+
];
|
|
12
|
+
/** INFANT, STILLBORN, CHILD in words (folded): the forms humanAge writes, and the Strom app's. */
|
|
13
|
+
const WORDS = {
|
|
14
|
+
infant: "INFANT", kojenec: "INFANT", saugling: "INFANT", niemowle: "INFANT",
|
|
15
|
+
stillborn: "STILLBORN", "mrtve narozene": "STILLBORN", "mrtve narozeny": "STILLBORN", totgeboren: "STILLBORN", "martwo urodzone": "STILLBORN",
|
|
16
|
+
child: "CHILD", dite: "CHILD", kind: "CHILD", dziecko: "CHILD",
|
|
17
|
+
};
|
|
18
|
+
/** Is `s` a valid GEDCOM 5.5.1 age? */
|
|
19
|
+
export function isGedcomAge(s) {
|
|
20
|
+
return KEYWORDS.has(s) || /^[<>]?(?:\d{1,3}y(?: \d{1,2}m)?(?: \d{1,3}d)?|\d{1,2}m(?: \d{1,3}d)?|\d{1,3}d)$/.test(s);
|
|
21
|
+
}
|
|
22
|
+
/** Words around an age that carry no amount: "aged", "ve věku", "aetatis", "asi". */
|
|
23
|
+
const FILLER = /^(aged|age|at|of|about|circa|ca|c|asi|cca|ve|věku|věk|stáří|v|im|alter|von|etwa|aetatis|aetat|około|w|wieku|environ|âgé|âgée|de|di|età|and|und|a|i|et|e|y)\.?$/i;
|
|
24
|
+
/** Normalise an age to GEDCOM form, or undefined when it cannot be read. */
|
|
25
|
+
export function normalizeAge(input) {
|
|
26
|
+
const raw = input.trim().replace(/\s+/g, " ");
|
|
27
|
+
if (!raw)
|
|
28
|
+
return undefined;
|
|
29
|
+
const up = raw.toUpperCase();
|
|
30
|
+
if (KEYWORDS.has(up))
|
|
31
|
+
return up;
|
|
32
|
+
// the keywords in words, as programs write them back ("kojenec", "mrtvě narozené", "Säugling")
|
|
33
|
+
const word = WORDS[raw.normalize("NFD").replace(/\p{M}/gu, "").toLowerCase()];
|
|
34
|
+
if (word)
|
|
35
|
+
return word;
|
|
36
|
+
if (isGedcomAge(raw))
|
|
37
|
+
return raw;
|
|
38
|
+
let rest = raw.replace(/,(?!\d)/g, " ").trim();
|
|
39
|
+
let sign = "";
|
|
40
|
+
if (/^[<>]/.test(rest)) {
|
|
41
|
+
sign = rest[0];
|
|
42
|
+
rest = rest.slice(1).trim();
|
|
43
|
+
}
|
|
44
|
+
if (/^\d{1,3}$/.test(rest))
|
|
45
|
+
return `${sign}${Number(rest)}y`; // "27" alone means years
|
|
46
|
+
const parts = { y: 0, m: 0, d: 0 };
|
|
47
|
+
let seen = false;
|
|
48
|
+
let pendingUnit; // "annorum 25": the unit before the number
|
|
49
|
+
const tokens = rest.match(/\d+(?:[.,]\d+)?|[^\d\s]+/g) ?? [];
|
|
50
|
+
const unitOf = (w) => (w ? UNITS.find(([re]) => re.test(w))?.[1] : undefined);
|
|
51
|
+
for (let i = 0; i < tokens.length; i++) {
|
|
52
|
+
const tok = tokens[i];
|
|
53
|
+
const n = Number(tok.replace(",", "."));
|
|
54
|
+
if (!Number.isFinite(n)) {
|
|
55
|
+
const u = unitOf(tok);
|
|
56
|
+
if (u)
|
|
57
|
+
pendingUnit = u;
|
|
58
|
+
else if (!FILLER.test(tok))
|
|
59
|
+
return undefined;
|
|
60
|
+
continue;
|
|
61
|
+
}
|
|
62
|
+
let unit = unitOf(tokens[i + 1]);
|
|
63
|
+
if (unit)
|
|
64
|
+
i++;
|
|
65
|
+
else
|
|
66
|
+
unit = pendingUnit ?? "y";
|
|
67
|
+
pendingUnit = undefined;
|
|
68
|
+
if (unit === "w")
|
|
69
|
+
parts.d += Math.round(n * 7);
|
|
70
|
+
else if (unit === "y" && !Number.isInteger(n)) {
|
|
71
|
+
parts.y += Math.floor(n);
|
|
72
|
+
parts.m += Math.round((n % 1) * 12);
|
|
73
|
+
}
|
|
74
|
+
else
|
|
75
|
+
parts[unit] += Math.round(n);
|
|
76
|
+
seen = true;
|
|
77
|
+
}
|
|
78
|
+
if (!seen)
|
|
79
|
+
return undefined;
|
|
80
|
+
const out = [parts.y ? `${parts.y}y` : "", parts.m ? `${parts.m}m` : "", parts.d ? `${parts.d}d` : ""].filter(Boolean).join(" ");
|
|
81
|
+
return out && isGedcomAge(`${sign}${out}`) ? `${sign}${out}` : undefined;
|
|
82
|
+
}
|
|
83
|
+
const HUMAN = {
|
|
84
|
+
en: { y: "y", m: "m", d: "d", INFANT: "infant", STILLBORN: "stillborn", CHILD: "child" },
|
|
85
|
+
cs: { y: "let", m: "měs.", d: "dní", INFANT: "kojenec", STILLBORN: "mrtvě narozené", CHILD: "dítě" },
|
|
86
|
+
de: { y: "J.", m: "Mon.", d: "Tage", INFANT: "Säugling", STILLBORN: "totgeboren", CHILD: "Kind" },
|
|
87
|
+
pl: { y: "lat", m: "mies.", d: "dni", INFANT: "niemowlę", STILLBORN: "martwo urodzone", CHILD: "dziecko" },
|
|
88
|
+
};
|
|
89
|
+
/** "27y 3m" → "27 let 3 měs." in the research language (for notes a reader sees). */
|
|
90
|
+
export function humanAge(age, lang) {
|
|
91
|
+
const t = HUMAN[lang] ?? HUMAN.en;
|
|
92
|
+
if (KEYWORDS.has(age))
|
|
93
|
+
return t[age];
|
|
94
|
+
return age.replace(/(\d+)([ymd])/g, (_, n, u) => `${n} ${t[u]}`);
|
|
95
|
+
}
|
|
@@ -0,0 +1,74 @@
|
|
|
1
|
+
// The desktop apps of the agents — the easiest way in for a person who has
|
|
2
|
+
// never used an AI agent: no terminal, the agent in a window of its own.
|
|
3
|
+
// strom finds them (by name, or by the link they open on Windows and Linux),
|
|
4
|
+
// and opens a conversation in one with a link: the tree folder and the first
|
|
5
|
+
// message filled in. The person confirms the folder and sends the message.
|
|
6
|
+
//
|
|
7
|
+
// Claude (Anthropic) claude://code/new?folder=…&q=… the Code tab, a local session
|
|
8
|
+
// ChatGPT / Codex codex://new?path=…&prompt=…
|
|
9
|
+
// OpenCode opencode://new?cwd=…&q=…
|
|
10
|
+
//
|
|
11
|
+
// Antigravity has an app too, but no known link that opens a conversation:
|
|
12
|
+
// strom talks to it in the terminal (agy).
|
|
13
|
+
import fs from "node:fs";
|
|
14
|
+
import path from "node:path";
|
|
15
|
+
import { spawnSync } from "node:child_process";
|
|
16
|
+
import { userHome } from "./paths.js";
|
|
17
|
+
import { AGENTS, findAgent, which } from "./which.js";
|
|
18
|
+
const q = encodeURIComponent;
|
|
19
|
+
export const DESKTOP_APPS = {
|
|
20
|
+
claude: { agent: "claude", name: "Claude", scheme: "claude", mac: ["Claude.app"], linux: "claude-desktop", link: (f, m) => `claude://code/new?folder=${q(f)}&q=${q(m)}` },
|
|
21
|
+
codex: { agent: "codex", name: "ChatGPT (Codex)", scheme: "codex", mac: ["Codex.app", "ChatGPT.app"], link: (f, m) => `codex://new?path=${q(f)}&prompt=${q(m)}` },
|
|
22
|
+
opencode: { agent: "opencode", name: "OpenCode", scheme: "opencode", mac: ["OpenCode.app"], link: (f, m) => `opencode://new?cwd=${q(f)}&q=${q(m)}` },
|
|
23
|
+
};
|
|
24
|
+
const seen = new Map();
|
|
25
|
+
/** Is this agent's desktop app on this computer? (STROM_APP_DIRS: folders to look in instead — tests.) */
|
|
26
|
+
export function hasDesktopApp(agent, env, platform = process.platform) {
|
|
27
|
+
const app = DESKTOP_APPS[agent];
|
|
28
|
+
if (!app)
|
|
29
|
+
return false;
|
|
30
|
+
const key = `${agent}|${platform}|${env.STROM_APP_DIRS ?? ""}|${env.HOME ?? env.USERPROFILE ?? ""}`;
|
|
31
|
+
const known = seen.get(key);
|
|
32
|
+
if (known !== undefined)
|
|
33
|
+
return known;
|
|
34
|
+
let found = false;
|
|
35
|
+
if (env.STROM_APP_DIRS !== undefined)
|
|
36
|
+
found = env.STROM_APP_DIRS.split(path.delimiter).some((d) => d && app.mac.some((n) => fs.existsSync(path.join(d, n))));
|
|
37
|
+
else if (platform === "darwin")
|
|
38
|
+
found = ["/Applications", path.join(userHome(env), "Applications")].some((d) => app.mac.some((n) => fs.existsSync(path.join(d, n))));
|
|
39
|
+
else if (platform === "win32") {
|
|
40
|
+
// An installed app registers its link for the user or for everyone.
|
|
41
|
+
found = ["HKCU", "HKLM"].some((root) => spawnSync("reg", ["query", `${root}\\Software\\Classes\\${app.scheme}`], { stdio: "ignore", windowsHide: true }).status === 0);
|
|
42
|
+
}
|
|
43
|
+
else {
|
|
44
|
+
if (app.linux && which(app.linux, env, platform))
|
|
45
|
+
found = true;
|
|
46
|
+
else {
|
|
47
|
+
const r = spawnSync("xdg-mime", ["query", "default", `x-scheme-handler/${app.scheme}`], { encoding: "utf8", env: env });
|
|
48
|
+
found = r.status === 0 && Boolean(r.stdout?.trim());
|
|
49
|
+
}
|
|
50
|
+
}
|
|
51
|
+
seen.set(key, found);
|
|
52
|
+
return found;
|
|
53
|
+
}
|
|
54
|
+
/** The agents strom can drive that are on this computer, in strom's order. */
|
|
55
|
+
export function agentsHere(env, platform = process.platform) {
|
|
56
|
+
return AGENTS.map((a) => ({ id: a.id, cli: Boolean(findAgent(a.command, env, platform)), app: hasDesktopApp(a.id, env, platform) })).filter((a) => a.cli || a.app);
|
|
57
|
+
}
|
|
58
|
+
/**
|
|
59
|
+
* The user's choice, else the app when it is there (the easiest way), else
|
|
60
|
+
* the terminal. An agent without an app, or an app strom cannot open, is the
|
|
61
|
+
* terminal whatever the choice.
|
|
62
|
+
*/
|
|
63
|
+
export function whereToTalk(agent, chosen, env) {
|
|
64
|
+
const here = agentsHere(env).find((a) => a.id === agent);
|
|
65
|
+
if (!here?.app)
|
|
66
|
+
return "terminal";
|
|
67
|
+
if (!here.cli)
|
|
68
|
+
return "app";
|
|
69
|
+
return chosen === "terminal" ? "terminal" : "app";
|
|
70
|
+
}
|
|
71
|
+
/** Is strom run by an agent inside a desktop app? (Claude Code says where it runs.) */
|
|
72
|
+
export function inDesktopApp(env) {
|
|
73
|
+
return /desktop/i.test(env.CLAUDE_CODE_ENTRYPOINT ?? "");
|
|
74
|
+
}
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
// Files shipped with strom (method pack, templates). They live in assets/
|
|
2
|
+
// next to src/ and dist/, so the same relative path works from both.
|
|
3
|
+
import fs from "node:fs";
|
|
4
|
+
import path from "node:path";
|
|
5
|
+
import { fileURLToPath } from "node:url";
|
|
6
|
+
const ROOT = path.resolve(path.dirname(fileURLToPath(import.meta.url)), "..", "..", "assets");
|
|
7
|
+
export function assetPath(...parts) {
|
|
8
|
+
return path.join(ROOT, ...parts);
|
|
9
|
+
}
|
|
10
|
+
const cache = new Map();
|
|
11
|
+
export function readAsset(...parts) {
|
|
12
|
+
const file = assetPath(...parts);
|
|
13
|
+
if (cache.has(file))
|
|
14
|
+
return cache.get(file);
|
|
15
|
+
try {
|
|
16
|
+
const text = fs.readFileSync(file, "utf8");
|
|
17
|
+
cache.set(file, text);
|
|
18
|
+
return text;
|
|
19
|
+
}
|
|
20
|
+
catch {
|
|
21
|
+
return undefined;
|
|
22
|
+
}
|
|
23
|
+
}
|
|
24
|
+
/** Method text for a task level: the core plus the level's own page (+ reading for work in books). */
|
|
25
|
+
export function methodFor(level) {
|
|
26
|
+
const parts = [readAsset("method", "core.md")];
|
|
27
|
+
if (level)
|
|
28
|
+
parts.push(readAsset("method", `${level}.md`));
|
|
29
|
+
if (level && ["link", "verify", "enrich"].includes(level))
|
|
30
|
+
parts.push(readAsset("method", "recording.md"));
|
|
31
|
+
if (level && ["link", "verify", "enrich", "intake"].includes(level))
|
|
32
|
+
parts.push(readAsset("method", "reading.md"));
|
|
33
|
+
return parts.filter(Boolean).join("\n");
|
|
34
|
+
}
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
// Keep the computer awake while agents work (the old workflow lost hours when
|
|
2
|
+
// a Mac fell asleep mid-session). Built-in OS tools only; ends with strom.
|
|
3
|
+
import { spawn } from "node:child_process";
|
|
4
|
+
import { which } from "./which.js";
|
|
5
|
+
export function keepAwake(env) {
|
|
6
|
+
let child;
|
|
7
|
+
try {
|
|
8
|
+
if (process.platform === "darwin")
|
|
9
|
+
child = spawn("caffeinate", ["-i", "-w", String(process.pid)], { stdio: "ignore" });
|
|
10
|
+
else if (process.platform === "win32")
|
|
11
|
+
child = spawn("powershell", [
|
|
12
|
+
"-NoProfile",
|
|
13
|
+
"-Command",
|
|
14
|
+
"$s='[DllImport(\"kernel32.dll\")] public static extern uint SetThreadExecutionState(uint f);';" +
|
|
15
|
+
"$t=Add-Type -MemberDefinition $s -Name Awake -Namespace Strom -PassThru;" +
|
|
16
|
+
"while($true){$t::SetThreadExecutionState(0x80000001)|Out-Null;Start-Sleep 50}",
|
|
17
|
+
], { stdio: "ignore", windowsHide: true });
|
|
18
|
+
else if (which("systemd-inhibit", env))
|
|
19
|
+
child = spawn("systemd-inhibit", ["--what=idle:sleep", "--who=strom", "--why=research session", "sleep", "infinity"], { stdio: "ignore" });
|
|
20
|
+
}
|
|
21
|
+
catch {
|
|
22
|
+
child = undefined;
|
|
23
|
+
}
|
|
24
|
+
child?.on("error", () => undefined);
|
|
25
|
+
return () => {
|
|
26
|
+
try {
|
|
27
|
+
child?.kill();
|
|
28
|
+
}
|
|
29
|
+
catch {
|
|
30
|
+
// already gone
|
|
31
|
+
}
|
|
32
|
+
};
|
|
33
|
+
}
|
|
@@ -0,0 +1,281 @@
|
|
|
1
|
+
// Images through the user's own browser. strom does not drive a browser — the
|
|
2
|
+
// agent does, with its browser tools (Claude in Chrome). strom plans what the
|
|
3
|
+
// browser fetches: the addresses the connector located, the file names, the
|
|
4
|
+
// times its limiter reserved. It gives the agent a script to run in a tab of
|
|
5
|
+
// the images' site, which fetches them one by one at that pace and saves each
|
|
6
|
+
// into the browser's downloads folder under its name. strom then takes the
|
|
7
|
+
// files over from there, checks them like any download and registers them with
|
|
8
|
+
// where they came from. What the browser got from the archive (a refusal) goes
|
|
9
|
+
// into the limiter's memory, as if strom had got it.
|
|
10
|
+
import fs from "node:fs";
|
|
11
|
+
import path from "node:path";
|
|
12
|
+
import crypto from "node:crypto";
|
|
13
|
+
import { userHome } from "./paths.js";
|
|
14
|
+
import { readJsonIfExists, writeJson } from "./json.js";
|
|
15
|
+
/** The longest one script waits between its images: a browser tool call must end in time. */
|
|
16
|
+
export const SCRIPT_MS = 40_000;
|
|
17
|
+
/** A plan is run within this time, or it is over (its times are long past): fetch again. */
|
|
18
|
+
export const PLAN_TTL_MS = 15 * 60_000;
|
|
19
|
+
/** A plan whose files never came is forgotten after this. */
|
|
20
|
+
export const KEEP_MS = 7 * 24 * 3600_000;
|
|
21
|
+
/** Browser tools of Claude in Chrome an agent may use for a connector's site. */
|
|
22
|
+
export const CHROME_ALLOW = [
|
|
23
|
+
"tabs_context_mcp",
|
|
24
|
+
"tabs_create_mcp",
|
|
25
|
+
"tabs_close_mcp",
|
|
26
|
+
"navigate",
|
|
27
|
+
"javascript_tool",
|
|
28
|
+
"read_page",
|
|
29
|
+
"get_page_text",
|
|
30
|
+
"find",
|
|
31
|
+
"form_input",
|
|
32
|
+
"computer",
|
|
33
|
+
"read_console_messages",
|
|
34
|
+
"read_network_requests",
|
|
35
|
+
"list_connected_browsers",
|
|
36
|
+
"select_browser",
|
|
37
|
+
"switch_browser",
|
|
38
|
+
"resize_window",
|
|
39
|
+
].map((t) => `mcp__claude-in-chrome__${t}`);
|
|
40
|
+
/** Never: a file of this computer uploaded to a web site, the user's own browser shortcuts, a batch the rules cannot see into. */
|
|
41
|
+
export const CHROME_DENY = ["file_upload", "upload_image", "shortcuts_execute", "gif_creator", "browser_batch"].map((t) => `mcp__claude-in-chrome__${t}`);
|
|
42
|
+
/** The permission rule of Claude in Chrome for one site. */
|
|
43
|
+
export const chromeDomain = (host) => `ClaudeInChromeDomain(${host.replace(/^\*\./, "").toLowerCase()})`;
|
|
44
|
+
/** The folder a browser saves downloads into by default: ~/Downloads, or what the desktop names it (Linux). */
|
|
45
|
+
export function downloadsDir(env, platform = process.platform) {
|
|
46
|
+
const home = userHome(env);
|
|
47
|
+
if (platform === "linux") {
|
|
48
|
+
const cfg = env.XDG_CONFIG_HOME || path.join(home, ".config");
|
|
49
|
+
try {
|
|
50
|
+
const m = /^XDG_DOWNLOAD_DIR="([^"]*)"/m.exec(fs.readFileSync(path.join(cfg, "user-dirs.dirs"), "utf8"));
|
|
51
|
+
if (m?.[1])
|
|
52
|
+
return m[1].replace(/^\$HOME/, home);
|
|
53
|
+
}
|
|
54
|
+
catch {
|
|
55
|
+
// no desktop settings: the usual name
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
return path.join(home, "Downloads");
|
|
59
|
+
}
|
|
60
|
+
const safe = (s) => s.normalize("NFC").replace(/[^\p{L}\p{M}\p{N}._-]+/gu, "_").replace(/^[._]+/, "") || "x";
|
|
61
|
+
/** The name a planned image is saved under: strom-<connector>-<book>-0009 (a part: …-0009-part). */
|
|
62
|
+
export function fileBase(connector, book, n, part) {
|
|
63
|
+
return `strom-${connector}-${safe(book)}-${String(n).padStart(4, "0")}${part ? "-part" : ""}`;
|
|
64
|
+
}
|
|
65
|
+
function plansDir(root) {
|
|
66
|
+
return path.join(root, ".strom", "browser");
|
|
67
|
+
}
|
|
68
|
+
export function savePlan(root, plan) {
|
|
69
|
+
writeJson(path.join(plansDir(root), `${plan.id}.json`), plan);
|
|
70
|
+
}
|
|
71
|
+
/** The plans of a tree still waiting for their files (of one connector), oldest first; forgotten ones removed. */
|
|
72
|
+
export function loadPlans(root, connector, now = Date.now()) {
|
|
73
|
+
const dir = plansDir(root);
|
|
74
|
+
if (!fs.existsSync(dir))
|
|
75
|
+
return [];
|
|
76
|
+
const out = [];
|
|
77
|
+
for (const f of fs.readdirSync(dir).filter((x) => x.endsWith(".json")).sort()) {
|
|
78
|
+
const p = readJsonIfExists(path.join(dir, f));
|
|
79
|
+
if (!p || !Array.isArray(p.items))
|
|
80
|
+
continue;
|
|
81
|
+
if ((!p.items.length && !p.pages?.length) || now - p.created > KEEP_MS) {
|
|
82
|
+
fs.rmSync(path.join(dir, f), { force: true });
|
|
83
|
+
continue;
|
|
84
|
+
}
|
|
85
|
+
if (!connector || p.connector === connector)
|
|
86
|
+
out.push(p);
|
|
87
|
+
}
|
|
88
|
+
return out;
|
|
89
|
+
}
|
|
90
|
+
export function removePlan(root, id) {
|
|
91
|
+
fs.rmSync(path.join(plansDir(root), `${id}.json`), { force: true });
|
|
92
|
+
}
|
|
93
|
+
/** The script the agent runs in a tab of the plan's site (top-level await). It returns one line: "strom-result 9:200:412918 10:403". */
|
|
94
|
+
export function planScript(plan) {
|
|
95
|
+
// the page of the image as the referrer, as the portal's own viewer sends it (a browser allows it on the same site)
|
|
96
|
+
const ref = (u) => (u && URL.canParse(u) && new URL(u).origin === plan.origin ? { ref: u } : {});
|
|
97
|
+
const data = { origin: plan.origin, open: plan.open, until: plan.until, gap: plan.gap, items: plan.items.map((i) => ({ n: i.n, src: i.src, file: i.file, at: i.at, ...ref(i.url) })) };
|
|
98
|
+
return `await (async () => {
|
|
99
|
+
const plan = ${JSON.stringify(data)};
|
|
100
|
+
if (location.origin !== plan.origin) return "strom: this tab is on " + location.origin + " — open " + plan.open + " first";
|
|
101
|
+
if (Date.now() > plan.until) return "strom: this plan is over — run strom fetch again for a new one";
|
|
102
|
+
const ext = { "image/jpeg": ".jpg", "image/png": ".png", "image/tiff": ".tif", "image/webp": ".webp", "image/gif": ".gif", "image/jp2": ".jp2" };
|
|
103
|
+
const out = [];
|
|
104
|
+
let last = 0;
|
|
105
|
+
for (const it of plan.items) {
|
|
106
|
+
const wait = Math.max(it.at, last + plan.gap) - Date.now();
|
|
107
|
+
if (wait > 0) await new Promise((r) => setTimeout(r, wait));
|
|
108
|
+
last = Date.now();
|
|
109
|
+
let res;
|
|
110
|
+
try {
|
|
111
|
+
res = await fetch(it.src, it.ref ? { credentials: "include", referrer: it.ref } : { credentials: "include" });
|
|
112
|
+
} catch (e) {
|
|
113
|
+
out.push(it.n + ":failed");
|
|
114
|
+
break;
|
|
115
|
+
}
|
|
116
|
+
if (!res.ok) {
|
|
117
|
+
out.push(it.n + ":" + res.status);
|
|
118
|
+
break;
|
|
119
|
+
}
|
|
120
|
+
const blob = await res.blob();
|
|
121
|
+
const a = document.createElement("a");
|
|
122
|
+
a.href = URL.createObjectURL(blob);
|
|
123
|
+
a.download = it.file + (ext[blob.type.split(";")[0].trim()] || ".jpg");
|
|
124
|
+
document.body.appendChild(a);
|
|
125
|
+
a.click();
|
|
126
|
+
a.remove();
|
|
127
|
+
setTimeout(() => URL.revokeObjectURL(a.href), 60000);
|
|
128
|
+
out.push(it.n + ":" + res.status + ":" + blob.size);
|
|
129
|
+
}
|
|
130
|
+
return "strom-result " + out.join(" ");
|
|
131
|
+
})()`;
|
|
132
|
+
}
|
|
133
|
+
/** The line the script returned: "strom-result 9:200:412918 10:403 11:failed". */
|
|
134
|
+
export function parseResult(s) {
|
|
135
|
+
const out = [];
|
|
136
|
+
for (const tok of s.replace(/^\s*["']?\s*strom-result\s*/i, "").split(/[\s,"']+/)) {
|
|
137
|
+
const m = /^(\d+|p[0-9a-f]{16}):(\d{3}|failed)(?::(\d+))?$/.exec(tok);
|
|
138
|
+
if (!m)
|
|
139
|
+
continue;
|
|
140
|
+
const page = m[1].startsWith("p") ? m[1].slice(1) : undefined;
|
|
141
|
+
out.push({ n: page ? 0 : Number(m[1]), ...(page ? { page } : {}), ...(m[2] === "failed" ? { failed: true } : { status: Number(m[2]) }), ...(m[3] ? { bytes: Number(m[3]) } : {}) });
|
|
142
|
+
}
|
|
143
|
+
return out;
|
|
144
|
+
}
|
|
145
|
+
const UNFINISHED = /\.(crdownload|part|download|tmp)$/i;
|
|
146
|
+
/**
|
|
147
|
+
* The downloaded file of a planned name: "<base>.jpg", or "<base> (1).jpg" when the
|
|
148
|
+
* browser had one of that name already — the newest. Names are compared composed
|
|
149
|
+
* (a folder may give them back decomposed); the path is the folder's own.
|
|
150
|
+
*/
|
|
151
|
+
export function findDownload(dir, base) {
|
|
152
|
+
let names;
|
|
153
|
+
try {
|
|
154
|
+
names = fs.readdirSync(dir);
|
|
155
|
+
}
|
|
156
|
+
catch {
|
|
157
|
+
return { others: [] };
|
|
158
|
+
}
|
|
159
|
+
const want = base.normalize("NFC");
|
|
160
|
+
const esc = want.replace(/[.*+?^${}()|[\]\\]/g, "\\$&");
|
|
161
|
+
const re = new RegExp(`^${esc}(?: \\(\\d+\\))?\\.[\\p{L}\\p{N}]{2,5}$`, "u");
|
|
162
|
+
const found = [];
|
|
163
|
+
let unfinished = false;
|
|
164
|
+
for (const name of names) {
|
|
165
|
+
const nfc = name.normalize("NFC");
|
|
166
|
+
if (!nfc.startsWith(want))
|
|
167
|
+
continue;
|
|
168
|
+
if (UNFINISHED.test(nfc)) {
|
|
169
|
+
unfinished = true;
|
|
170
|
+
continue;
|
|
171
|
+
}
|
|
172
|
+
if (!re.test(nfc))
|
|
173
|
+
continue;
|
|
174
|
+
const file = path.join(dir, name);
|
|
175
|
+
found.push({ file, at: fs.statSync(file).mtimeMs });
|
|
176
|
+
}
|
|
177
|
+
found.sort((a, b) => b.at - a.at);
|
|
178
|
+
const [best, ...rest] = found;
|
|
179
|
+
return { ...(best ? { file: best.file } : {}), others: rest.map((f) => f.file), ...(!best && unfinished ? { unfinished: true } : {}) };
|
|
180
|
+
}
|
|
181
|
+
// ── pages through the browser (a portal behind a bot check) ─────────────────
|
|
182
|
+
/** A request's key: the same request is the same page (and the connector asks the same, run after run). */
|
|
183
|
+
export function pageKey(r) {
|
|
184
|
+
return crypto.createHash("sha256").update(`${r.method} ${r.url}\n${r.body ?? ""}`).digest("hex").slice(0, 16);
|
|
185
|
+
}
|
|
186
|
+
/** A page kept this long answers the connector again; then the browser asks for it anew. */
|
|
187
|
+
export const PAGE_TTL_MS = 24 * 3600_000;
|
|
188
|
+
const pagesDir = (root, connector) => path.join(root, ".strom", "browser", "pages", connector);
|
|
189
|
+
export function loadPage(root, connector, key, now = Date.now()) {
|
|
190
|
+
const p = readJsonIfExists(path.join(pagesDir(root, connector), `${key}.json`));
|
|
191
|
+
if (!p || now - p.at > PAGE_TTL_MS)
|
|
192
|
+
return undefined;
|
|
193
|
+
return { status: p.status, contentType: p.type, headers: p.headers, url: p.finalUrl, body: Buffer.from(p.body, "base64") };
|
|
194
|
+
}
|
|
195
|
+
export function savePage(root, connector, req, got, now = Date.now()) {
|
|
196
|
+
const kept = { key: req.key, method: req.method, url: req.url, at: now, status: got.status, type: got.contentType, headers: got.headers, finalUrl: got.url, body: got.body.toString("base64") };
|
|
197
|
+
writeJson(path.join(pagesDir(root, connector), `${req.key}.json`), kept);
|
|
198
|
+
}
|
|
199
|
+
/** What a browser saved for a page: the file the page script wrote, or why it is not one. */
|
|
200
|
+
export function readSavedPage(file) {
|
|
201
|
+
let o;
|
|
202
|
+
try {
|
|
203
|
+
o = JSON.parse(fs.readFileSync(file, "utf8"));
|
|
204
|
+
}
|
|
205
|
+
catch {
|
|
206
|
+
return "not a page saved by strom's script (not JSON)";
|
|
207
|
+
}
|
|
208
|
+
if (o.strom !== "page" || typeof o.status !== "number" || typeof o.body !== "string")
|
|
209
|
+
return "not a page saved by strom's script";
|
|
210
|
+
const headers = o.headers && typeof o.headers === "object" ? Object.fromEntries(Object.entries(o.headers).map(([k, v]) => [k.toLowerCase(), String(v)])) : {};
|
|
211
|
+
return { status: o.status, contentType: String(o.type ?? ""), headers, url: String(o.url ?? ""), body: Buffer.from(o.body, "base64") };
|
|
212
|
+
}
|
|
213
|
+
/**
|
|
214
|
+
* A page that is a check whether a person is there (Imperva/Incapsula, Cloudflare, a captcha), not
|
|
215
|
+
* the page asked for: its name, or undefined. The services put their scripts into every page they
|
|
216
|
+
* guard, so a script alone is no check: a check is a small page with little else (found live).
|
|
217
|
+
*/
|
|
218
|
+
export function botCheck(body, headers = {}) {
|
|
219
|
+
const buf = typeof body === "string" ? Buffer.from(body) : body;
|
|
220
|
+
const text = buf.subarray(0, 64 * 1024).toString("latin1").toLowerCase();
|
|
221
|
+
const small = buf.length < 16 * 1024;
|
|
222
|
+
// what a reader sees of it: the text without tags and scripts
|
|
223
|
+
const words = text.replace(/<script[\s\S]*?<\/script>|<style[\s\S]*?<\/style>|<[^>]+>/g, " ").replace(/\s+/g, " ").trim().length;
|
|
224
|
+
const bare = small && words < 400;
|
|
225
|
+
const header = (name) => Object.entries(headers).find(([k]) => k.toLowerCase() === name)?.[1];
|
|
226
|
+
if (/incapsula incident id|request unsuccessful\. incapsula/.test(text) || (bare && /_incapsula_resource/.test(text)))
|
|
227
|
+
return "Imperva (Incapsula)";
|
|
228
|
+
if (/^challenge/i.test(header("cf-mitigated") ?? "") || /<title>(just a moment\.\.\.|attention required! \| cloudflare)<\/title>/.test(text) || (bare && /cf-challenge|cf_chl_|challenge-platform/.test(text)))
|
|
229
|
+
return "Cloudflare";
|
|
230
|
+
if (bare && /ddos-guard/.test(text))
|
|
231
|
+
return "DDoS-Guard";
|
|
232
|
+
if (bare && /g-recaptcha|h-captcha|hcaptcha\.com|recaptcha\/api\.js/.test(text))
|
|
233
|
+
return "a captcha";
|
|
234
|
+
return undefined;
|
|
235
|
+
}
|
|
236
|
+
/** Headers a page's own script may send (a browser sets the rest itself: cookies, the user agent). */
|
|
237
|
+
const PAGE_HEADERS = new Set(["accept", "accept-language", "content-type", "x-requested-with"]);
|
|
238
|
+
/** The script that asks the plan's pages in the tab (top-level await): each saved as <file>.json; it returns "strom-result p<key>:200:5120". */
|
|
239
|
+
export function pageScript(plan) {
|
|
240
|
+
const pages = (plan.pages ?? []).map((p) => {
|
|
241
|
+
const h = Object.fromEntries(Object.entries(p.headers).filter(([k]) => PAGE_HEADERS.has(k.toLowerCase())));
|
|
242
|
+
const ref = Object.entries(p.headers).find(([k]) => k.toLowerCase() === "referer")?.[1];
|
|
243
|
+
return { key: p.key, method: p.method, url: p.url, file: p.file, at: p.at, headers: h, ...(p.body !== undefined ? { body: p.body } : {}), ...(ref && URL.canParse(ref) && new URL(ref).origin === plan.origin ? { ref } : {}) };
|
|
244
|
+
});
|
|
245
|
+
const data = { origin: plan.origin, open: plan.open, until: plan.until, gap: plan.gap, pages };
|
|
246
|
+
return `await (async () => {
|
|
247
|
+
const plan = ${JSON.stringify(data)};
|
|
248
|
+
if (location.origin !== plan.origin) return "strom: this tab is on " + location.origin + " — open " + plan.open + " first";
|
|
249
|
+
if (Date.now() > plan.until) return "strom: this plan is over — run the strom command again for a new one";
|
|
250
|
+
const out = [];
|
|
251
|
+
let last = 0;
|
|
252
|
+
for (const p of plan.pages) {
|
|
253
|
+
const wait = Math.max(p.at, last + plan.gap) - Date.now();
|
|
254
|
+
if (wait > 0) await new Promise((r) => setTimeout(r, wait));
|
|
255
|
+
last = Date.now();
|
|
256
|
+
let res;
|
|
257
|
+
try {
|
|
258
|
+
res = await fetch(p.url, { method: p.method, headers: p.headers, credentials: "include", ...(p.body !== undefined ? { body: p.body } : {}), ...(p.ref ? { referrer: p.ref } : {}) });
|
|
259
|
+
} catch (e) {
|
|
260
|
+
out.push("p" + p.key + ":failed");
|
|
261
|
+
break;
|
|
262
|
+
}
|
|
263
|
+
const bytes = new Uint8Array(await res.arrayBuffer());
|
|
264
|
+
let bin = "";
|
|
265
|
+
for (let i = 0; i < bytes.length; i += 32768) bin += String.fromCharCode.apply(null, bytes.subarray(i, i + 32768));
|
|
266
|
+
const headers = {};
|
|
267
|
+
res.headers.forEach((v, k) => (headers[k] = v));
|
|
268
|
+
const saved = { strom: "page", status: res.status, type: res.headers.get("content-type") || "", url: res.url, headers, body: btoa(bin) };
|
|
269
|
+
const a = document.createElement("a");
|
|
270
|
+
a.href = URL.createObjectURL(new Blob([JSON.stringify(saved)], { type: "application/json" }));
|
|
271
|
+
a.download = p.file + ".json";
|
|
272
|
+
document.body.appendChild(a);
|
|
273
|
+
a.click();
|
|
274
|
+
a.remove();
|
|
275
|
+
setTimeout(() => URL.revokeObjectURL(a.href), 60000);
|
|
276
|
+
out.push("p" + p.key + ":" + res.status + ":" + bytes.length);
|
|
277
|
+
if (res.status === 401 || res.status === 403 || res.status === 429) break;
|
|
278
|
+
}
|
|
279
|
+
return "strom-result " + out.join(" ");
|
|
280
|
+
})()`;
|
|
281
|
+
}
|
|
@@ -0,0 +1,48 @@
|
|
|
1
|
+
// Page ↔ image calibration of a record set.
|
|
2
|
+
/** Fit page = a·image + b to numeric calibration points. */
|
|
3
|
+
export function fitCalibration(points) {
|
|
4
|
+
const num = points.map((p) => ({ x: p.image, y: Number.parseInt(p.page, 10) })).filter((p) => Number.isFinite(p.y));
|
|
5
|
+
if (num.length < 2)
|
|
6
|
+
return undefined;
|
|
7
|
+
const n = num.length;
|
|
8
|
+
const mx = num.reduce((s, p) => s + p.x, 0) / n;
|
|
9
|
+
const my = num.reduce((s, p) => s + p.y, 0) / n;
|
|
10
|
+
const sxx = num.reduce((s, p) => s + (p.x - mx) ** 2, 0);
|
|
11
|
+
if (sxx === 0)
|
|
12
|
+
return undefined;
|
|
13
|
+
const a = num.reduce((s, p) => s + (p.x - mx) * (p.y - my), 0) / sxx;
|
|
14
|
+
const b = my - a * mx;
|
|
15
|
+
const maxError = Math.max(...num.map((p) => Math.abs(a * p.x + b - p.y)));
|
|
16
|
+
return { a, b, maxError };
|
|
17
|
+
}
|
|
18
|
+
export function calibrationLine(b) {
|
|
19
|
+
if (b.calibration.length === 0)
|
|
20
|
+
return undefined;
|
|
21
|
+
const fit = fitCalibration(b.calibration);
|
|
22
|
+
const pts = b.calibration.map((c) => `${c.image}→${c.page}`).join(" ");
|
|
23
|
+
if (!fit)
|
|
24
|
+
return `calibration ${pts}`;
|
|
25
|
+
const formula = `page ≈ ${+fit.a.toFixed(3)}·image ${fit.b >= 0 ? "+" : "−"} ${Math.abs(+fit.b.toFixed(2))}`;
|
|
26
|
+
const quality = fit.maxError < 0.51 ? "consistent" : `INCONSISTENT (off by up to ${fit.maxError.toFixed(1)} — the book is not linear; measure, do not compute)`;
|
|
27
|
+
return `calibration ${pts} · ${formula} · ${quality}`;
|
|
28
|
+
}
|
|
29
|
+
/** The page on an image: measured, or computed from a consistent calibration ("≈"). */
|
|
30
|
+
export function pageOf(b, image) {
|
|
31
|
+
const exact = b.calibration.find((c) => c.image === image);
|
|
32
|
+
if (exact)
|
|
33
|
+
return exact.page;
|
|
34
|
+
const fit = fitCalibration(b.calibration);
|
|
35
|
+
if (!fit || fit.maxError >= 0.51)
|
|
36
|
+
return undefined;
|
|
37
|
+
return `≈${Math.round(fit.a * image + fit.b)}`;
|
|
38
|
+
}
|
|
39
|
+
/** The image a page is on, from a consistent calibration (undefined when the book is not linear). */
|
|
40
|
+
export function imageOf(b, page) {
|
|
41
|
+
const exact = b.calibration.find((c) => Number.parseInt(c.page, 10) === page);
|
|
42
|
+
if (exact)
|
|
43
|
+
return exact.image;
|
|
44
|
+
const fit = fitCalibration(b.calibration);
|
|
45
|
+
if (!fit || fit.maxError >= 0.51 || fit.a === 0)
|
|
46
|
+
return undefined;
|
|
47
|
+
return Math.round((page - fit.b) / fit.a);
|
|
48
|
+
}
|