strom-research 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (130) hide show
  1. package/LICENSE +373 -0
  2. package/README.md +142 -0
  3. package/assets/lang/cs.json +302 -0
  4. package/assets/lang/de.json +302 -0
  5. package/assets/method/core.md +43 -0
  6. package/assets/method/enrich.md +11 -0
  7. package/assets/method/intake.md +30 -0
  8. package/assets/method/link.md +28 -0
  9. package/assets/method/locate.md +28 -0
  10. package/assets/method/narrate.md +13 -0
  11. package/assets/method/reading.md +62 -0
  12. package/assets/method/recording.md +59 -0
  13. package/assets/method/request.md +10 -0
  14. package/assets/method/verify.md +17 -0
  15. package/assets/plugins/README.md +23 -0
  16. package/assets/plugins/connectors/DISCOVERY.md +159 -0
  17. package/assets/plugins/connectors/README.md +376 -0
  18. package/assets/plugins/connectors/sdk.ts +168 -0
  19. package/assets/plugins/connectors/template.ts +38 -0
  20. package/assets/plugins/gitignore +4 -0
  21. package/dist/agents/files.js +313 -0
  22. package/dist/agents/global.js +257 -0
  23. package/dist/agents/launch.js +36 -0
  24. package/dist/agents/profiles.js +95 -0
  25. package/dist/brief/brief.js +345 -0
  26. package/dist/cli/commit.js +44 -0
  27. package/dist/cli/context.js +311 -0
  28. package/dist/cli/execute.js +154 -0
  29. package/dist/cli/fixes.js +78 -0
  30. package/dist/cli/format.js +53 -0
  31. package/dist/cli/help.js +59 -0
  32. package/dist/cli/main.js +152 -0
  33. package/dist/cli/menu.js +212 -0
  34. package/dist/cli/registry.js +96 -0
  35. package/dist/cli/ui.js +266 -0
  36. package/dist/cli/wizard.js +142 -0
  37. package/dist/cli.js +14 -0
  38. package/dist/commands/analysis.js +622 -0
  39. package/dist/commands/batch.js +181 -0
  40. package/dist/commands/checks.js +153 -0
  41. package/dist/commands/connectors.js +1377 -0
  42. package/dist/commands/guide.js +160 -0
  43. package/dist/commands/index.js +19 -0
  44. package/dist/commands/intake.js +234 -0
  45. package/dist/commands/media.js +406 -0
  46. package/dist/commands/meta.js +195 -0
  47. package/dist/commands/output.js +117 -0
  48. package/dist/commands/people.js +664 -0
  49. package/dist/commands/read.js +199 -0
  50. package/dist/commands/research.js +139 -0
  51. package/dist/commands/session.js +605 -0
  52. package/dist/commands/setup.js +465 -0
  53. package/dist/commands/sources.js +634 -0
  54. package/dist/commands/start.js +383 -0
  55. package/dist/commands/story.js +75 -0
  56. package/dist/commands/tasks.js +436 -0
  57. package/dist/commands/trees.js +128 -0
  58. package/dist/core/actions.js +852 -0
  59. package/dist/core/age.js +95 -0
  60. package/dist/core/apps.js +74 -0
  61. package/dist/core/assets.js +34 -0
  62. package/dist/core/awake.js +33 -0
  63. package/dist/core/browser.js +281 -0
  64. package/dist/core/calibration.js +48 -0
  65. package/dist/core/check.js +112 -0
  66. package/dist/core/chromium.js +88 -0
  67. package/dist/core/config.js +348 -0
  68. package/dist/core/connector.js +811 -0
  69. package/dist/core/deps.js +73 -0
  70. package/dist/core/dialog.js +61 -0
  71. package/dist/core/errors.js +89 -0
  72. package/dist/core/evidence.js +58 -0
  73. package/dist/core/frontier.js +219 -0
  74. package/dist/core/gdate.js +77 -0
  75. package/dist/core/git.js +300 -0
  76. package/dist/core/guard.js +124 -0
  77. package/dist/core/http2.js +76 -0
  78. package/dist/core/import.js +541 -0
  79. package/dist/core/install.js +28 -0
  80. package/dist/core/integrity.js +219 -0
  81. package/dist/core/json.js +87 -0
  82. package/dist/core/lang.js +70 -0
  83. package/dist/core/live.js +244 -0
  84. package/dist/core/lock.js +112 -0
  85. package/dist/core/logins.js +67 -0
  86. package/dist/core/media.js +223 -0
  87. package/dist/core/model.js +101 -0
  88. package/dist/core/net.js +366 -0
  89. package/dist/core/open.js +29 -0
  90. package/dist/core/paths.js +84 -0
  91. package/dist/core/people.js +283 -0
  92. package/dist/core/phrases.js +85 -0
  93. package/dist/core/queue.js +113 -0
  94. package/dist/core/reader.js +76 -0
  95. package/dist/core/records.js +105 -0
  96. package/dist/core/roles.js +30 -0
  97. package/dist/core/schema.js +261 -0
  98. package/dist/core/seal.js +77 -0
  99. package/dist/core/self.js +40 -0
  100. package/dist/core/session.js +155 -0
  101. package/dist/core/shortcut.js +90 -0
  102. package/dist/core/stories.js +61 -0
  103. package/dist/core/stromapp.js +138 -0
  104. package/dist/core/text.js +104 -0
  105. package/dist/core/tree.js +507 -0
  106. package/dist/core/uninstall.js +128 -0
  107. package/dist/core/update.js +193 -0
  108. package/dist/core/validate.js +260 -0
  109. package/dist/core/views.js +164 -0
  110. package/dist/core/which.js +51 -0
  111. package/dist/core/workers.js +42 -0
  112. package/dist/gedcom/export.js +454 -0
  113. package/dist/gedcom/labels.js +103 -0
  114. package/dist/gedcom/lines.js +91 -0
  115. package/dist/gedcom/parse.js +53 -0
  116. package/dist/gedcom/validate.js +183 -0
  117. package/dist/image/image.js +223 -0
  118. package/dist/image/index.js +114 -0
  119. package/dist/image/jpeg-decode.js +552 -0
  120. package/dist/image/jpeg-encode.js +254 -0
  121. package/dist/image/png.js +241 -0
  122. package/dist/runners/antigravity.js +70 -0
  123. package/dist/runners/claude.js +179 -0
  124. package/dist/runners/codex.js +45 -0
  125. package/dist/runners/index.js +13 -0
  126. package/dist/runners/jsonl.js +86 -0
  127. package/dist/runners/opencode.js +50 -0
  128. package/dist/runners/runner.js +63 -0
  129. package/dist/runners/script.js +58 -0
  130. package/package.json +44 -0
@@ -0,0 +1,376 @@
1
+ # Connectors — interface 1
2
+
3
+ A connector downloads from one archive portal: it finds the books (registers)
4
+ that cover a place, describes a book, and fetches its images. strom runs it.
5
+ The connector never reaches the network, other programs or files outside its
6
+ work folder by itself: it asks strom, which paces every request, keeps to the
7
+ connector's hosts, writes the files and checks that an image is an image.
8
+
9
+ This page is the whole contract. It is **version 1 and it does not change**:
10
+ a connector written for it keeps working. (A different contract would get a
11
+ new number, and strom would still run version 1.)
12
+
13
+ ## The folder
14
+
15
+ connectors/<name>/ the folder's name is the connector's name:
16
+ lowercase letters a–z, digits and dashes
17
+ connector.json the manifest (below)
18
+ connector.ts the program (any language; TypeScript is the usual one)
19
+ sdk.ts the SDK for TypeScript (strom connector new copies it in)
20
+ README.md how the portal is mapped, for whoever keeps it working
21
+
22
+ **Install** = copy the folder here. **Remove** = delete it. `strom connector new
23
+ <name> --url <portal>` starts one with everything in place. Folders whose name
24
+ starts with `.` or `_` are ignored (`_old-version/`).
25
+
26
+ ## connector.json
27
+
28
+ {
29
+ "interface": 1,
30
+ "title": "Example State Archive — digital reading room",
31
+ "run": ["node", "connector.ts"],
32
+ "hosts": ["digi.example.org", "iiif.example.org"],
33
+ "can": ["find", "list", "fetch", "part", "locate"],
34
+ "routes": ["direct", "browser"],
35
+ "policy": {
36
+ "automation": "allowed",
37
+ "terms": "https://digi.example.org/terms",
38
+ "termsSummary": "Images may be downloaded for private research …",
39
+ "robots": "Crawl-delay: 5",
40
+ "officialExport": "IIIF manifests for every book",
41
+ "pace": { "minIntervalMs": 5000 }
42
+ },
43
+ "login": {
44
+ "about": "An account of the portal: its members see the scans at full size",
45
+ "url": "https://digi.example.org/register",
46
+ "fields": { "user": "User name or e-mail", "password": "Password" }
47
+ }
48
+ }
49
+
50
+ - `interface`: `1`.
51
+ - `title`: the archive or portal, as people call it.
52
+ - `run`: the program and its arguments, started in the connector's folder.
53
+ `"node"` is the Node that runs strom (it runs TypeScript as it is).
54
+ - `hosts`: every host it contacts. `"example.org"` includes its subdomains.
55
+ strom refuses requests anywhere else, and each host needs the user's consent.
56
+ - `can`: what it does: any of `find`, `list`, `fetch`, `part`, `locate`.
57
+ - `routes` (optional): how its images may come, the usual one first:
58
+ - `direct`: strom fetches them (`fetch`, `part`). The default: `["direct"]`.
59
+ - `browser`: the user's own browser fetches them (section 5). It needs
60
+ `locate` in `can`. `["browser"]` alone: only through the browser — for a
61
+ portal that lets in only a browser, such as a login with a second factor.
62
+
63
+ The user chooses which one is used (`strom connector use <name> --via
64
+ browser`), and can switch back.
65
+ - `browser` (optional):
66
+ - `open`: the page a tab opens first when the browser fetches — where the
67
+ user logs in to the portal. It must be on the images' site. Without it,
68
+ the tab opens a light page of that site.
69
+ - `pages`: `true` for a portal that answers a real browser only (a bot check
70
+ such as Imperva or Cloudflare): every request of the connector goes through
71
+ the user's browser (section 5). It needs `"routes": ["browser"]` alone and
72
+ no `login`: the user logs in in their own browser.
73
+ - `policy`: what the portal allows, found out before the connector was written.
74
+ - `automation`: one of
75
+ - `allowed`: the terms allow it or say nothing against it;
76
+ - `manual`: the terms forbid automated download. strom fetches no images
77
+ through it, but it may still find books and give their links;
78
+ - `unknown`.
79
+ - `terms` (a URL), `termsSummary`, `robots` and `officialExport`: shown to
80
+ the user when they decide.
81
+ - `pace` (optional): slower than strom's default, never faster.
82
+ - `minIntervalMs`: time between two requests to a host. The default and
83
+ the minimum is 2000.
84
+ - `perHour`: requests to a host in an hour. The default and the maximum is
85
+ 400.
86
+ - `login` (optional): the portal gives more to users who log in, and the
87
+ connector can use the user's own account.
88
+ - `about`: what an account gives, in a sentence the user reads.
89
+ - `url` (optional): where to get an account.
90
+ - `fields`: what the user types in, each with its question. A field named
91
+ `user` or `email` is typed in the open; every other one (a password, a
92
+ key) is hidden, and kept out of the answers the connector reads.
93
+ - `required` (optional): `true` when it cannot work without one.
94
+
95
+ The user saves their login in their own terminal (`strom login <name>`),
96
+ never an agent. It stays on their computer. See section 4.
97
+ - `version` (optional): the connector's own version.
98
+
99
+ ## The conversation
100
+
101
+ strom starts the program and talks to it in JSON lines: one JSON object per
102
+ line, on its standard input and output (UTF-8). Standard error is the
103
+ program's own; its last lines are shown when it fails.
104
+
105
+ ### 1. What strom asks: the first line on stdin
106
+
107
+ {"interface":1,"cmd":"find","place":"Dolní Lhota","years":"1780-1850"}
108
+ {"interface":1,"cmd":"list","book":"4711"}
109
+ {"interface":1,"cmd":"fetch","book":"4711","images":[40,41,42]}
110
+ {"interface":1,"cmd":"part","book":"4711","image":40,"region":{"x":0.5,"y":0.25,"w":0.5,"h":0.4}}
111
+ {"interface":1,"cmd":"locate","book":"4711","images":[40,41]}
112
+
113
+ - `find`: the books that cover a place. `years` ("from-to") is optional.
114
+ - `list`: one book: its title and how many images it has.
115
+ - `fetch`: these images of the book, in this order. Images are numbered as
116
+ the portal counts them, from 1.
117
+ - `part`: a part of one image, as sharp as the portal gives it — asked for
118
+ when the whole image is too small to read an entry. `region` is where the
119
+ part is in the whole image, in fractions of it (0–1) from its top left
120
+ corner. Answer with one `image` line.
121
+ - `locate`: where these images are, for the user's browser to fetch them
122
+ (section 5). Download nothing: answer with a `located` line for each. With
123
+ `region`, where that part of the one image is.
124
+ - `book` is the connector's own ID of a book: whatever it gave as `id` in
125
+ `find`.
126
+ - `login`: `true` when the user saved a login for this connector. The values
127
+ are not in it: see section 4.
128
+
129
+ ### 2. What the connector says: lines on stdout
130
+
131
+ {"id":1,"http":{"url":"https://digi.example.org/book/4711"}}
132
+ {"book":{"id":"4711","title":"Dolní Lhota N 1784–1820","callNumber":"17","years":"1784-1820","kinds":["baptism"],"places":["Dolní Lhota"],"url":"https://digi.example.org/book/4711","images":109}}
133
+ {"image":{"n":40,"file":"s0040.jpg","url":"https://digi.example.org/book/4711/40","page":"fol. 19v"}}
134
+ {"located":{"n":40,"src":"https://iiif.example.org/4711/40/full/max/0/default.jpg","url":"https://digi.example.org/book/4711/40"}}
135
+ {"log":"reading the catalogue"}
136
+ {"error":"book 4711 has no images online"}
137
+ {"done":true}
138
+
139
+ - `http`: a request (section 3).
140
+ - `book`: one for every book found (`find`), or the book described (`list`).
141
+ Only `title` is required; give everything else the portal says. `kinds`
142
+ holds baptism, marriage, burial, index and so on.
143
+ - `image`: one image fetched.
144
+ - `n`: its number in the book.
145
+ - `file`: the name it was saved under (the `save` of its request).
146
+ - `url`: where it is on the portal: the page a person would open, or the
147
+ image's own address.
148
+ - `page` (optional): a folio or page label.
149
+ - `region` (`part` only, optional): the part it got, when the portal cut it
150
+ differently from the part asked for.
151
+ - `tiles`, `width`, `height` (optional): the portal gives the image in tiles
152
+ only (Zoomify, DeepZoom, IIIF tiles). Save each tile with `save` and list
153
+ them: `"tiles":[{"file":"t40-0-0.jpg","x":0,"y":0},…]`, where `x`, `y` is
154
+ the tile's top left corner in pixels of the image, and `width` × `height`
155
+ its size. strom puts them together into `file` (a JPEG) and removes the
156
+ tiles. Tiles may overlap; a gap — a tile left out — ends the run. For a
157
+ part, the tiles make the part, and `region` says where it is.
158
+
159
+ {"image":{"n":40,"file":"s0040.jpg","tiles":[{"file":"t40-0-0.jpg","x":0,"y":0},{"file":"t40-1-0.jpg","x":540,"y":0}],"width":1080,"height":540,"url":"https://digi.example.org/book/4711/40"}}
160
+
161
+ strom checks that the file is an image and not cut short, and that it is not
162
+ a stand-in: an image smaller than 400 px on its longer side (a part: 64 px) is
163
+ a placeholder or a thumbnail, and so is the same file as another image of the
164
+ run. If it is one of these, the run ends.
165
+ - `located` (`locate` only): where one image is.
166
+ - `n`: its number in the book.
167
+ - `src`: the image's own address, on one of `hosts` — the one request that
168
+ gives the image, as `fetch` would ask for it.
169
+ - `url`: the page a person would open (kept with the image).
170
+ - `page`, `region`: as in `image`.
171
+ - `log`: a line for the user.
172
+ - `error`: the connector cannot go on. The run ends.
173
+ - `done`: finished. strom closes the connector's stdin once every request is
174
+ answered, and the program then exits.
175
+
176
+ A line that is not JSON counts as a log line.
177
+
178
+ ### 3. Requests: the only way to the network
179
+
180
+ {"id":1,"http":{"url":"https://digi.example.org/book/4711"}}
181
+ {"id":2,"http":{"url":"https://iiif.example.org/4711/40/full/max/0/default.jpg","save":"s0040.jpg","headers":{"Referer":"https://digi.example.org/"}}}
182
+ {"id":3,"http":{"url":"https://digi.example.org/search","method":"POST","body":"place=Doln%C3%AD+Lhota","headers":{"Content-Type":"application/x-www-form-urlencoded"}}}
183
+ {"id":4,"http":{"url":"https://digi.example.org/search","form":{"place":"Dolní Lhota"}}}
184
+
185
+ - `id`: a number the connector chooses. The answer carries it back.
186
+ - `url`: http or https, on one of `hosts`.
187
+ - `method`: `GET` (the default), `POST` or `HEAD`.
188
+ - `body`: a string, sent with a POST.
189
+ - `form` (instead of `body`): the fields of a form, as an object. strom
190
+ sends them with POST, as `application/x-www-form-urlencoded`.
191
+ - `headers`: request headers such as Referer, Accept, Content-Type,
192
+ X-Requested-With or a Cookie of its own. strom sets `User-Agent` itself:
193
+ who is asking is not the connector's to change.
194
+ - `save`: a file name. strom writes the answer's body into the work folder
195
+ under this name; use it for images and anything big. Without `save`, the
196
+ body comes back as text (at most 5 MB).
197
+
198
+ strom follows redirects itself, and each target must be on `hosts`. It keeps
199
+ the cookies that servers set during a run and sends them back, as a browser
200
+ does: a session, or the result of a form. Every run starts without cookies.
201
+
202
+ The answer is one line on stdin:
203
+
204
+ {"id":1,"status":200,"type":"text/html; charset=utf-8","headers":{"content-type":"text/html; charset=utf-8"},"url":"https://digi.example.org/book/4711","text":"<html>…"}
205
+ {"id":2,"status":200,"type":"image/jpeg","headers":{"content-type":"image/jpeg"},"url":"https://iiif.example.org/4711/40/full/max/0/default.jpg","file":"/…/s0040.jpg","bytes":1843221,"width":4200,"height":3100}
206
+ {"id":3,"error":"refused","message":"digi.example.org answered 403 — …"}
207
+
208
+ - `status`: the HTTP status. A 404 is an answer, not an error.
209
+ - `url`: the final address, after redirects.
210
+ - `headers`: the answer's headers, in lowercase.
211
+ - `width`, `height`: of an image saved with `save` (JPEG or PNG), read from
212
+ its header. A thumbnail or a placeholder shows here before the connector
213
+ gives it as the image: it can ask elsewhere instead.
214
+ - `error`: strom got no answer, or will not ask. After an error the run ends,
215
+ and strom stops the connector. The errors are:
216
+ - `host`: the host is not on the list;
217
+ - `refused`: the archive said no (401 or 403), or asked twice to slow down
218
+ (429);
219
+ - `blocked`: the archive refused earlier and is left alone for now;
220
+ - `cap`: the hourly cap is reached;
221
+ - `silent`: no answer. The server is down, or it blocks this IP;
222
+ - `http`: the server keeps failing, or the request is not valid;
223
+ - `too-big`: text over 5 MB (ask with `save`);
224
+ - `login`: a request with a login value (section 4) that strom will not send.
225
+
226
+ strom answers requests one at a time, in order, and paces them:
227
+ - at least 2 s apart for each host, and at most 400 an hour;
228
+ - shared by everything on this computer;
229
+ - it waits by itself when an archive asks it to (Retry-After).
230
+
231
+ ### 4. The user's login
232
+
233
+ A value of a `form` field or of a header may be the user's login instead of
234
+ text: `{"login":"<field>"}`, a field of `login.fields` in `connector.json`.
235
+
236
+ {"id":5,"http":{"url":"https://digi.example.org/login","form":{"name":{"login":"user"},"pass":{"login":"password"},"token":"a41f"}}}
237
+ {"id":6,"http":{"url":"https://api.example.org/scan/40","save":"s0040.jpg","headers":{"X-Api-Key":{"login":"key"}}}}
238
+
239
+ - strom puts in what the user saved. The connector never sees it.
240
+ - It goes only over https, and only to the hosts the login was saved for.
241
+ It does not follow a redirect to another address.
242
+ - The hidden values (every field but `user` and `email`) are taken out of
243
+ every answer the connector reads, written as `[login]`.
244
+ - The session a login starts lives in the cookies of the run: every run logs
245
+ in again.
246
+ - Without a saved login, such a request is refused (`login`). Use it only when
247
+ the first line says `"login":true`.
248
+
249
+ A login the user does not have is a technical measure: never get round it.
250
+
251
+ ### 5. Through the user's browser
252
+
253
+ With the route `browser`, the images come through the user's own browser,
254
+ where their login to the portal lives. strom never runs a browser; the user's
255
+ agent does, with its browser tools. For `strom fetch`:
256
+
257
+ 1. strom asks the connector to `locate` the images. Its pages still go
258
+ through strom, paced, as in any run.
259
+ 2. strom's limiter reserves a time for each request of the browser, in the
260
+ state every strom process shares.
261
+ 3. The agent opens a tab on the images' site (`browser.open`) and runs a
262
+ script strom gives it. The script fetches each `src` at its time and saves
263
+ it into the browser's downloads folder under a name strom chose.
264
+ 4. `strom fetch <name> --take` takes the files over: checked like any
265
+ download, registered with where they came from. When the archive refused
266
+ the browser, strom leaves it alone as if it had refused strom.
267
+
268
+ So `src` must be on the same site as the page the tab opens, or allow that
269
+ page to fetch it. The browser sends the user's cookies of that site: their
270
+ login counts. The connector itself stays as it is: it never sees the browser,
271
+ and the browser never runs its code.
272
+
273
+ **Pages through the browser** (`browser.pages`): a portal behind a bot check
274
+ answers strom with the check, never with its pages. Then every request of the
275
+ connector — the search, a book's page, what `locate` reads — is made by the
276
+ user's browser too:
277
+
278
+ 1. strom runs the connector. A request it has no page for ends the run, and
279
+ strom plans it: a time in the limiter, a script for the tab.
280
+ 2. The script asks for the page in the tab and saves it into the downloads
281
+ folder. `strom fetch <name> --take` takes it over and runs the same command
282
+ again: the connector gets the pages the browser got, in order, and goes on
283
+ until it needs the next one. The pages are kept a day.
284
+ 3. The check whether a person is there is passed by the user in their own
285
+ browser — never by strom, the agent or the connector. When the site gives
286
+ its check instead of a page, strom does not take it and says so.
287
+
288
+ The connector is written as any other: `get()` and `post()`, one after the
289
+ other. Only the headers a page's own script may send reach the portal (Accept,
290
+ Content-Type, X-Requested-With; a Referer becomes the tab's referrer); the
291
+ browser sends its cookies and its own user agent.
292
+
293
+ ## Where it may write
294
+
295
+ The environment variable `STROM_CONNECTOR_WORKDIR` names the work folder.
296
+ - Files saved with `save` are there.
297
+ - The connector may write there itself. Tiles are put together by strom
298
+ (`tiles` of `image`).
299
+ - `image.file` names a file in it.
300
+
301
+ The environment holds nothing else of the user's shell. A Node connector runs
302
+ under Node's permission model:
303
+ - it may read its own folder and the work folder;
304
+ - it may write only into the work folder;
305
+ - it may start no other programs.
306
+
307
+ ## Rules
308
+
309
+ - **Never reach the network yourself.** Do not use `fetch()`, http or net
310
+ modules, `requests`, `urllib`, `socket`, curl or other programs. strom
311
+ checks the code and warns the user. A connector that does this needs the
312
+ user's consent again after every change.
313
+ - **Prefer what the archive offers**: a download of a whole book, IIIF, an
314
+ API. Where the portal allows it, make one request per image (a full-size
315
+ IIIF image, the image's own link) rather than hundreds of tiles.
316
+ - **Tiles only**: fetch the whole image at the level nearest 2000 px on its
317
+ longer side (what a viewer shows of a page), and for `part` the tiles of
318
+ that region at full size — tens of requests, not hundreds.
319
+ - **Never get round a technical measure**: logins you do not have, captchas,
320
+ or tokens meant to stop scripts. A check that a person is there (a captcha,
321
+ a bot check) is passed by the user in their own browser, never by code; after
322
+ it, the requests go at a person's pace through that browser
323
+ (`browser.pages`), only for what the research needs.
324
+ - **Fetch only what was asked for**, in order.
325
+
326
+ ## The SDK (TypeScript)
327
+
328
+ `sdk.ts` does the conversation:
329
+
330
+ import { request, get, post, login, book, image, located, log, done, fail } from "./sdk.ts";
331
+
332
+ const req = await request(); // {cmd:"find",place,years?} · {cmd:"list",book} · {cmd:"fetch",book,images} · {cmd:"part",book,image,region} · {cmd:"locate",book,images,region?}
333
+ const page = await get(url); // {status, text, type, headers, url}
334
+ const img = await get(url, { save: "s0040.jpg", headers: { Referer: "https://digi.example.org/" } }); // {status, file, bytes, …}
335
+ const hits = await post(url, { place: "Dolní Lhota" }); // a form, sent as application/x-www-form-urlencoded
336
+ if (req.login) await post(loginUrl, { name: login("user"), pass: login("password") }); // strom puts the user's login in
337
+ book({ id: "4711", title: "…", images: 109 });
338
+ image(40, "s0040.jpg", url); // for part too: the image it is a part of
339
+ imageFromTiles(40, "s0040.jpg", [{ file: "t40-0-0.jpg", x: 0, y: 0 }, …], { width: 2158, height: 1616 }, url); // tiles only: strom puts them together
340
+ located(40, src, pageUrl); // locate: where image 40 is, nothing downloaded
341
+ log("…");
342
+ done();
343
+
344
+ When strom refuses a request, the run is over: strom stops the connector.
345
+
346
+ The newest `sdk.ts` is next to this file. A connector written with an older
347
+ one copies it over its own to use what came later (`imageFromTiles`, the
348
+ `width` and `height` of a saved image).
349
+
350
+ ## Testing and consent
351
+
352
+ strom connector test <name> --find "<place>"
353
+ strom connector test <name> --list <book>
354
+ strom connector test <name> --fetch <book> --images 1-2
355
+ strom connector test <name> --fetch <book> --images 2 --crop 0.5,0,0.5,0.5 (part)
356
+ strom connector test <name> --locate <book> --images 1-2 (locate)
357
+
358
+ A test makes at most 10 requests (`--max`: up to 50, for an image in tiles);
359
+ its files go to `<name>/.test/`, and strom shows each image's size. While you
360
+ map a portal, `strom connector probe <name> <url>` makes one request through
361
+ strom and saves the answer in `<name>/.test/probe/`, to read: this is how the
362
+ pages and scripts of a JavaScript application are read. Probes keep the cookies
363
+ servers set, as one visit in a browser does (`--fresh` starts without them).
364
+ `strom connector grep <name> <text>` searches what was saved and shows each hit
365
+ with the text round it.
366
+
367
+ A connector runs as soon as it is in the folder. Two things need the user's
368
+ consent, given in their own terminal (`strom allow connector <name>`, or when
369
+ they run a test or a fetch themselves, where strom asks right away):
370
+ - code that reaches the network itself, past strom: always, and again after
371
+ every change of it;
372
+ - every connector and each of its hosts, when the user asked to be asked first
373
+ (`strom config set connectors.consent on`).
374
+
375
+ When a consent is missing, strom answers with exit code 4 and the command for
376
+ the user. An agent never gives it.
@@ -0,0 +1,168 @@
1
+ // The strom connector SDK (interface 1) — the conversation with strom over
2
+ // stdin/stdout, in JSON lines. A connector never touches the network itself:
3
+ // get() and post() ask strom, which paces the request, checks the host against
4
+ // connector.json, keeps the cookies of the run and — with `save` — writes the
5
+ // file into the work folder. The user's login goes in as login("password"):
6
+ // strom puts in the value, the connector never sees it. The contract is
7
+ // ../README.md. Needs nothing but Node.
8
+
9
+ import readline from "node:readline";
10
+
11
+ /** Where a part is in the whole image, in fractions of it (0–1) from its top left corner. */
12
+ export interface Region {
13
+ x: number;
14
+ y: number;
15
+ w: number;
16
+ h: number;
17
+ }
18
+
19
+ /** `login`: the user saved a login for this connector (strom login <name>). */
20
+ export type Request = { interface: 1; login?: boolean } & (
21
+ | { cmd: "find"; place: string; years?: string }
22
+ | { cmd: "list"; book: string }
23
+ | { cmd: "fetch"; book: string; images: number[] }
24
+ | { cmd: "part"; book: string; image: number; region: Region }
25
+ | { cmd: "locate"; book: string; images: number[]; region?: Region }
26
+ );
27
+
28
+ /** A field of the user's login ("user", "password" … of connector.json → login.fields): strom puts in its value. */
29
+ export interface LoginValue {
30
+ login: string;
31
+ }
32
+
33
+ /** The user's login in a form field or a header: post(url, { pass: login("password") }). */
34
+ export function login(field: string): LoginValue {
35
+ return { login: field };
36
+ }
37
+
38
+ export interface Got {
39
+ status: number;
40
+ /** The body, when not saved (text, HTML, JSON). */
41
+ text?: string;
42
+ /** Where strom saved it, with `save`. */
43
+ file?: string;
44
+ type?: string;
45
+ bytes?: number;
46
+ /** An image saved: its size from its header (JPEG, PNG) — a thumbnail or a placeholder shows here before you give it as the image. */
47
+ width?: number;
48
+ height?: number;
49
+ /** The answer's headers, in lowercase. */
50
+ headers?: Record<string, string>;
51
+ /** The final URL, after redirects. */
52
+ url?: string;
53
+ }
54
+
55
+ export interface HttpOptions {
56
+ /** Request headers: Referer, Accept, X-Requested-With … (strom sets User-Agent); a value may be login("key"). */
57
+ headers?: Record<string, string | LoginValue>;
58
+ /** A file name in the work folder: strom writes the body there (images, big files). */
59
+ save?: string;
60
+ }
61
+
62
+ /** strom would not ask, or got no answer: a refusal, the hourly cap, a host not in connector.json … The run is over. */
63
+ export class Refused extends Error {
64
+ readonly reason: string;
65
+ constructor(reason: string, message: string) {
66
+ super(message);
67
+ this.reason = reason;
68
+ }
69
+ }
70
+
71
+ const lines = readline.createInterface({ input: process.stdin });
72
+ const waiting = new Map<number, (answer: Record<string, unknown>) => void>();
73
+ let first: ((r: Request) => void) | undefined;
74
+ const firstLine = new Promise<Request>((resolve) => (first = resolve));
75
+ let nextId = 1;
76
+
77
+ lines.on("line", (line) => {
78
+ if (!line.trim()) return;
79
+ const msg = JSON.parse(line) as Record<string, unknown>;
80
+ if (first) {
81
+ const f = first;
82
+ first = undefined;
83
+ return f(msg as unknown as Request);
84
+ }
85
+ const done = waiting.get(msg.id as number);
86
+ if (done) {
87
+ waiting.delete(msg.id as number);
88
+ done(msg);
89
+ }
90
+ });
91
+
92
+ function send(o: unknown): void {
93
+ process.stdout.write(JSON.stringify(o) + "\n");
94
+ }
95
+
96
+ /** What strom asks for: find books, list a book, fetch its images or a part of one, or say where they are. */
97
+ export function request(): Promise<Request> {
98
+ return firstLine;
99
+ }
100
+
101
+ /** Any request through strom: GET (default), POST or HEAD; a form is sent with POST. */
102
+ export function http(url: string, opts: HttpOptions & { method?: "GET" | "POST" | "HEAD"; body?: string; form?: Record<string, string | LoginValue> } = {}): Promise<Got> {
103
+ const id = nextId++;
104
+ send({ id, http: { url, ...opts } });
105
+ return new Promise((resolve, reject) =>
106
+ waiting.set(id, (a) => (a.error ? reject(new Refused(String(a.error), String(a.message ?? a.error))) : resolve(a as unknown as Got))),
107
+ );
108
+ }
109
+
110
+ /** GET a URL through strom. */
111
+ export function get(url: string, opts: HttpOptions = {}): Promise<Got> {
112
+ return http(url, opts);
113
+ }
114
+
115
+ /** POST through strom: a form (an object, sent URL-encoded; a value may be login("password")) or a body of your own (set its Content-Type). */
116
+ export function post(url: string, body: Record<string, string | LoginValue> | string, opts: HttpOptions = {}): Promise<Got> {
117
+ return typeof body === "string" ? http(url, { ...opts, method: "POST", body }) : http(url, { ...opts, method: "POST", form: body });
118
+ }
119
+
120
+ /** One image fetched — or a part of it (cmd part): its number in the book, the file saved with `save`, where it is on the portal. */
121
+ export function image(n: number, file: string, url: string, page?: string, region?: Region): void {
122
+ send({ image: { n, file, url, ...(page ? { page } : {}), ...(region ? { region } : {}) } });
123
+ }
124
+
125
+ /** One tile of an image the portal gives in tiles only, saved with `save`: where its top left corner is, in pixels of the whole image. */
126
+ export interface Tile {
127
+ file: string;
128
+ x: number;
129
+ y: number;
130
+ }
131
+
132
+ /**
133
+ * An image the portal gives in tiles only (Zoomify, DeepZoom …): strom puts the
134
+ * tiles together into `file` (a JPEG), width × height pixels, and checks it as
135
+ * any image. For a part (cmd part), width × height is the part and `region`
136
+ * where it is in the whole image. One request for the whole image is better
137
+ * where the portal has one: tiles are many requests.
138
+ */
139
+ export function imageFromTiles(n: number, file: string, tiles: Tile[], size: { width: number; height: number }, url: string, page?: string, region?: Region): void {
140
+ send({ image: { n, file, tiles, width: size.width, height: size.height, url, ...(page ? { page } : {}), ...(region ? { region } : {}) } });
141
+ }
142
+
143
+ /** Where an image is (cmd locate), for the user's browser to fetch: its own address — one request gives it — and the page a person opens. */
144
+ export function located(n: number, src: string, url?: string, page?: string, region?: Region): void {
145
+ send({ located: { n, src, ...(url ? { url } : {}), ...(page ? { page } : {}), ...(region ? { region } : {}) } });
146
+ }
147
+
148
+ /** One book found (find), or the book described (list). */
149
+ export function book(b: { id?: string; title: string; callNumber?: string; years?: string; kinds?: string[]; places?: string[]; url?: string; images?: number }): void {
150
+ send({ book: b });
151
+ }
152
+
153
+ export function log(message: string): void {
154
+ send({ log: message });
155
+ }
156
+
157
+ /** The end: strom closes the connector when it has answered everything. */
158
+ export function done(): void {
159
+ send({ done: true });
160
+ lines.close();
161
+ }
162
+
163
+ /** Something the connector cannot do (a book it does not know, a page it cannot read). */
164
+ export function fail(message: string): never {
165
+ send({ error: message });
166
+ lines.close();
167
+ process.exit(1);
168
+ }
@@ -0,0 +1,38 @@
1
+ // A strom connector for __TITLE__ (__URL__) — interface 1 (../README.md).
2
+ // strom runs it (strom fetch, strom connector test); it reaches the portal
3
+ // only through get() and post() — never with fetch() or a network library.
4
+ // Read DISCOVERY.md first: what the portal allows decides what this may do.
5
+
6
+ import { book, done, fail, get, image, log, request } from "./sdk.ts";
7
+
8
+ const req = await request();
9
+
10
+ if (req.cmd === "find") {
11
+ // The books that cover a place (and years): the portal's catalogue search.
12
+ // TODO: get (or post, for a search form) the catalogue for req.place, read it, and for each book:
13
+ // book({ id: "…", title: "…", callNumber: "…", years: "1784-1820", kinds: ["baptism"], places: ["…"], url: "…", images: 70 });
14
+ log(`find ${req.place} ${req.years ?? ""}: not written yet`);
15
+ } else if (req.cmd === "list") {
16
+ // One book: its title and number of images, from the portal's page for req.book.
17
+ log(`list ${req.book}: not written yet`);
18
+ } else if (req.cmd === "fetch") {
19
+ // The images asked for, one by one, in order — one request per image where the portal allows it.
20
+ for (const n of req.images) {
21
+ // TODO: the URL of image n of book req.book — the portal's own full-image link, or IIIF.
22
+ const url = `__URL__/${encodeURIComponent(req.book)}/${n}`;
23
+ const file = `s${String(n).padStart(4, "0")}.jpg`;
24
+ const got = await get(url, { save: file });
25
+ if (got.status !== 200) fail(`image ${n}: HTTP ${got.status}`);
26
+ image(n, file, url);
27
+ }
28
+ } else if (req.cmd === "part") {
29
+ // Only with "part" in connector.json → can: a part of one image, sharper than the whole image
30
+ // (IIIF: the region of the original; one request). req.region is in fractions of the whole image.
31
+ fail("part: not written yet");
32
+ } else if (req.cmd === "locate") {
33
+ // Only with "locate" in can and "browser" in routes: where the images are, for the user's browser
34
+ // to fetch — the same address fetch would ask for (a part: req.region), nothing downloaded here.
35
+ // located(n, url, "<the page a person opens>");
36
+ fail("locate: not written yet");
37
+ }
38
+ done();
@@ -0,0 +1,4 @@
1
+ # Plugins are programs on this computer, allowed by you: never part of a
2
+ # repository. Only strom's own README files are left in.
3
+ /*/*
4
+ !/*/README.md