strom-research 1.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (130) hide show
  1. package/LICENSE +373 -0
  2. package/README.md +142 -0
  3. package/assets/lang/cs.json +302 -0
  4. package/assets/lang/de.json +302 -0
  5. package/assets/method/core.md +43 -0
  6. package/assets/method/enrich.md +11 -0
  7. package/assets/method/intake.md +30 -0
  8. package/assets/method/link.md +28 -0
  9. package/assets/method/locate.md +28 -0
  10. package/assets/method/narrate.md +13 -0
  11. package/assets/method/reading.md +62 -0
  12. package/assets/method/recording.md +59 -0
  13. package/assets/method/request.md +10 -0
  14. package/assets/method/verify.md +17 -0
  15. package/assets/plugins/README.md +23 -0
  16. package/assets/plugins/connectors/DISCOVERY.md +159 -0
  17. package/assets/plugins/connectors/README.md +376 -0
  18. package/assets/plugins/connectors/sdk.ts +168 -0
  19. package/assets/plugins/connectors/template.ts +38 -0
  20. package/assets/plugins/gitignore +4 -0
  21. package/dist/agents/files.js +313 -0
  22. package/dist/agents/global.js +257 -0
  23. package/dist/agents/launch.js +36 -0
  24. package/dist/agents/profiles.js +95 -0
  25. package/dist/brief/brief.js +345 -0
  26. package/dist/cli/commit.js +44 -0
  27. package/dist/cli/context.js +311 -0
  28. package/dist/cli/execute.js +154 -0
  29. package/dist/cli/fixes.js +78 -0
  30. package/dist/cli/format.js +53 -0
  31. package/dist/cli/help.js +59 -0
  32. package/dist/cli/main.js +152 -0
  33. package/dist/cli/menu.js +212 -0
  34. package/dist/cli/registry.js +96 -0
  35. package/dist/cli/ui.js +266 -0
  36. package/dist/cli/wizard.js +142 -0
  37. package/dist/cli.js +14 -0
  38. package/dist/commands/analysis.js +622 -0
  39. package/dist/commands/batch.js +181 -0
  40. package/dist/commands/checks.js +153 -0
  41. package/dist/commands/connectors.js +1377 -0
  42. package/dist/commands/guide.js +160 -0
  43. package/dist/commands/index.js +19 -0
  44. package/dist/commands/intake.js +234 -0
  45. package/dist/commands/media.js +406 -0
  46. package/dist/commands/meta.js +195 -0
  47. package/dist/commands/output.js +117 -0
  48. package/dist/commands/people.js +664 -0
  49. package/dist/commands/read.js +199 -0
  50. package/dist/commands/research.js +139 -0
  51. package/dist/commands/session.js +605 -0
  52. package/dist/commands/setup.js +465 -0
  53. package/dist/commands/sources.js +634 -0
  54. package/dist/commands/start.js +383 -0
  55. package/dist/commands/story.js +75 -0
  56. package/dist/commands/tasks.js +436 -0
  57. package/dist/commands/trees.js +128 -0
  58. package/dist/core/actions.js +852 -0
  59. package/dist/core/age.js +95 -0
  60. package/dist/core/apps.js +74 -0
  61. package/dist/core/assets.js +34 -0
  62. package/dist/core/awake.js +33 -0
  63. package/dist/core/browser.js +281 -0
  64. package/dist/core/calibration.js +48 -0
  65. package/dist/core/check.js +112 -0
  66. package/dist/core/chromium.js +88 -0
  67. package/dist/core/config.js +348 -0
  68. package/dist/core/connector.js +811 -0
  69. package/dist/core/deps.js +73 -0
  70. package/dist/core/dialog.js +61 -0
  71. package/dist/core/errors.js +89 -0
  72. package/dist/core/evidence.js +58 -0
  73. package/dist/core/frontier.js +219 -0
  74. package/dist/core/gdate.js +77 -0
  75. package/dist/core/git.js +300 -0
  76. package/dist/core/guard.js +124 -0
  77. package/dist/core/http2.js +76 -0
  78. package/dist/core/import.js +541 -0
  79. package/dist/core/install.js +28 -0
  80. package/dist/core/integrity.js +219 -0
  81. package/dist/core/json.js +87 -0
  82. package/dist/core/lang.js +70 -0
  83. package/dist/core/live.js +244 -0
  84. package/dist/core/lock.js +112 -0
  85. package/dist/core/logins.js +67 -0
  86. package/dist/core/media.js +223 -0
  87. package/dist/core/model.js +101 -0
  88. package/dist/core/net.js +366 -0
  89. package/dist/core/open.js +29 -0
  90. package/dist/core/paths.js +84 -0
  91. package/dist/core/people.js +283 -0
  92. package/dist/core/phrases.js +85 -0
  93. package/dist/core/queue.js +113 -0
  94. package/dist/core/reader.js +76 -0
  95. package/dist/core/records.js +105 -0
  96. package/dist/core/roles.js +30 -0
  97. package/dist/core/schema.js +261 -0
  98. package/dist/core/seal.js +77 -0
  99. package/dist/core/self.js +40 -0
  100. package/dist/core/session.js +155 -0
  101. package/dist/core/shortcut.js +90 -0
  102. package/dist/core/stories.js +61 -0
  103. package/dist/core/stromapp.js +138 -0
  104. package/dist/core/text.js +104 -0
  105. package/dist/core/tree.js +507 -0
  106. package/dist/core/uninstall.js +128 -0
  107. package/dist/core/update.js +193 -0
  108. package/dist/core/validate.js +260 -0
  109. package/dist/core/views.js +164 -0
  110. package/dist/core/which.js +51 -0
  111. package/dist/core/workers.js +42 -0
  112. package/dist/gedcom/export.js +454 -0
  113. package/dist/gedcom/labels.js +103 -0
  114. package/dist/gedcom/lines.js +91 -0
  115. package/dist/gedcom/parse.js +53 -0
  116. package/dist/gedcom/validate.js +183 -0
  117. package/dist/image/image.js +223 -0
  118. package/dist/image/index.js +114 -0
  119. package/dist/image/jpeg-decode.js +552 -0
  120. package/dist/image/jpeg-encode.js +254 -0
  121. package/dist/image/png.js +241 -0
  122. package/dist/runners/antigravity.js +70 -0
  123. package/dist/runners/claude.js +179 -0
  124. package/dist/runners/codex.js +45 -0
  125. package/dist/runners/index.js +13 -0
  126. package/dist/runners/jsonl.js +86 -0
  127. package/dist/runners/opencode.js +50 -0
  128. package/dist/runners/runner.js +63 -0
  129. package/dist/runners/script.js +58 -0
  130. package/package.json +44 -0
@@ -0,0 +1,43 @@
1
+ # Method: the core
2
+
3
+ You work to the Genealogical Proof Standard: search thoroughly, cite every
4
+ fact, analyse and correlate the evidence, resolve conflicts, and write the
5
+ conclusion so the next researcher can follow it.
6
+
7
+ - **Certainty is explicit.** `proven` — you read the original record yourself
8
+ and it states the fact directly. `probable` — good evidence, but indirect or a
9
+ single reading of hard handwriting. `possible` — weak evidence. `lead` — a
10
+ family story, a family tree, an index or a guess: something to check, never a
11
+ fact. Family trees and memories are leads, always.
12
+ - **A namesake is not your person.** Identify by year, place, house, occupation,
13
+ spouse and parents — the mother's maiden name decides most cases. Two entries
14
+ that give one name different parents (another mother) are **two people** —
15
+ even in the same house, with the same father's name. Never explain the
16
+ difference away as a clerk's error, never decide by majority. Your person is
17
+ who the entry about your person's own family names; the others are separate
18
+ persons, and "the same woman?" is a hypothesis (`strom hypothesis add`) with
19
+ what would decide it.
20
+ - **Negative results are results.** Every search is recorded (`strom search add
21
+ … --result negative`), with exactly what was covered. The most expensive
22
+ mistake is to search the same book twice.
23
+ - **Check the premise first.** Before opening anything: `strom searched <where>`.
24
+ - **Extract everything the first time** you open a record: names, ages, house
25
+ numbers, occupations, godparents, witnesses, midwife, remarks in the margin.
26
+ Opening the same page again later costs more than writing it down now.
27
+ - **Nothing is deleted.** A wrong fact is retracted with a reason; a changed
28
+ fact is edited with a reason. Contradictions become conflicts, competing
29
+ explanations become hypotheses — never silently pick one.
30
+ - **Write as you go.** Record each finding as soon as you have it. Your context
31
+ can be cut or summarised at any moment; what is only in the conversation is
32
+ lost, what is in strom is not.
33
+ - **Lessons belong where they apply.** A quirk of a register (its calibration,
34
+ two years per page, a hand that writes 7 like 1) goes to `strom lesson add
35
+ --on B…`, so whoever opens that book next sees it.
36
+ - **Stay within the task.** New questions become new tasks (`strom task add`,
37
+ with where and done-when), not detours. A new task whose images are not here
38
+ gets them now: through the archive's connector (built first when it has
39
+ none), or — where the archive allows only that — it waits at once for what
40
+ the user should download (`strom task wait T… --images B…:<numbers> --on
41
+ "…"`); a later session would only find that out.
42
+ - **Hand over cleanly.** Close with a summary of what was proven, what was
43
+ searched in vain, and the next cheapest step.
@@ -0,0 +1,11 @@
1
+ # Method: enriching a person
2
+
3
+ The person is placed in the tree; now give them a life: occupations, every
4
+ house they lived in, siblings, godparents, military service, land, events in
5
+ the village. Each detail is a fact with a citation.
6
+
7
+ - Work outward from records already found: the same book usually holds the
8
+ siblings and the next generation.
9
+ - Keep the certainty honest: a house number from a baptism of a sibling is
10
+ evidence about the family at that date, not about the person's whole life.
11
+ - Record what you learn about the place and the books as lessons.
@@ -0,0 +1,30 @@
1
+ # Method: processing an input
2
+
3
+ An input is material the user gave you: scans, photos, documents, notes, a
4
+ family tree. Your job is to turn it into evidence without inventing anything.
5
+
6
+ 1. `strom input show I…` — read the file (images and PDFs directly, text is
7
+ shown). Identify who it is about and what kind of document it is.
8
+ 2. **What is it?** An original civil or church document (birth, marriage or
9
+ death certificate, extract from a register, military papers) is a **source**
10
+ and can prove facts: `strom source add … --input I… --form original
11
+ --information primary` (primary if written at the time of the event).
12
+ A photo, letter, family tree or someone's memory is authored or derivative:
13
+ everything taken from it stays a **lead**. What the user tells you is
14
+ `strom source add "<what it is>" --kind family-memory --form authored
15
+ --information secondary --input I…`; a family tree is `--kind family-tree`.
16
+ Many facts from one input: one `strom batch --file notes/<input>.txt`
17
+ (try it with `--dry-run` first).
18
+ 3. Record every person and every fact it states, with a citation to the source
19
+ (`--cite S… --locator "…"`). Use the words of the record in `--quote`.
20
+ 4. Put names and places as they are written; normalise only dates.
21
+ 5. For imported family trees: check for duplicates and impossible dates, but do
22
+ not "correct" them from memory — they are leads. A person of the tree who is
23
+ already researched here (the brief lists the likely ones) is merged INTO the
24
+ researched person, which keeps its evidence: `strom person merge <ours>
25
+ <imported> --reason "…"` — their parents first (`strom family merge`). What
26
+ the tree says differently stays a lead or becomes a conflict, never a fact.
27
+ 6. Create the next tasks: where would the records that prove these leads be?
28
+ (`locate` if the archive or book is unknown, `link` if it is known.)
29
+ 7. Close the task with what came out of it (`strom task done T… --result "…"`;
30
+ its input is marked processed with it).
@@ -0,0 +1,28 @@
1
+ # Method: proving a link (parents, marriage)
2
+
3
+ 1. **Premise:** `strom searched B… --years …` and `strom task show T…` — if the
4
+ range was already covered completely, do not repeat it; say so and close.
5
+ 2. **Index before book.** Use an index to find the page, then read the entry in
6
+ the register itself. An index entry alone is a lead — and so is its absence:
7
+ an index is a copy, written by another hand. A near spelling of the surname
8
+ in the right house or years (one letter off, a similar-looking letter of the
9
+ old script) is a candidate to check in the register by its parents, not a
10
+ different family: ask for that page (`strom task wait`) before you rule it
11
+ out.
12
+ 3. **Calibrate** the book before counting pages: measure image ↔ page at two
13
+ distant places (`strom recordset calibrate B… --point 95=189 --point 300=611`).
14
+ Never compute a page you have not measured in a book that is not linear.
15
+ 4. **Identify** the person by year, place, house number, parents and, above
16
+ all, the mother's maiden name. Two people of the same name in one village
17
+ are common.
18
+ 5. **Cheapest route:** a death or burial entry gives age and often parents; a
19
+ marriage entry names both sets of parents and the ages of the couple; the age
20
+ column is an independent check of a birth year.
21
+ 6. **Extract the whole entry** (see core). Godparents and witnesses are often
22
+ relatives and prove the next link.
23
+ 7. **Record** (see "recording an entry"): source (with transcript in the
24
+ original language), facts with citations, names, family links, the search
25
+ (found or negative), lessons about the book. `proven` only for what you read
26
+ yourself and the entry states directly.
27
+ 8. Done when the entry is found and recorded — or the agreed range was searched
28
+ completely without it (negative search recorded).
@@ -0,0 +1,28 @@
1
+ # Method: locating records
2
+
3
+ Before anything can be proven you need to know where the records are.
4
+
5
+ 1. Place and time decide the jurisdiction. Record it: `strom place add …` and
6
+ `strom place jurisdiction L… --kind parish|civil|manor|district --from --to`.
7
+ Borders, parishes and names changed; a village may belong to different
8
+ parishes in different centuries.
9
+ 2. Find the holding archive or portal and the concrete book or collection:
10
+ `strom repo add …` (record its terms of use and whether automated download
11
+ is allowed), `strom recordset add … --places --years --kinds --access --url`.
12
+ A connector the user allowed for the portal finds its books for you:
13
+ `strom fetch <connector> --find "<place>" --years 1780-1850`.
14
+ 3. Respect the terms of every archive. Never scrape a portal whose terms forbid
15
+ it; if records are only on site or on request, the task becomes a `request`.
16
+ 4. Indexes, catalogues and online trees tell you where to look — they are not
17
+ evidence of the fact itself. Use a catalogue as a person would: a search or
18
+ the page of one book — never walk its record numbers one after another
19
+ (archives block that). A catalogue you cannot read (an app, a login): ask
20
+ the user — the archive, the village, the years, the kind of book — with
21
+ `strom task wait T… --on "…"`.
22
+ 5. Record where you looked, also what gave nothing: `strom search add "<what>"
23
+ --method web|catalog --result found|negative --task T…` — the next session
24
+ must not search the same catalogues again.
25
+ 6. Done when a record set with access (URL or call number) exists and the next
26
+ `link` task points at it: `strom task edit T… --where B…` for a task that
27
+ was written before the book was known, `strom task add … --where B…` for a
28
+ new one.
@@ -0,0 +1,13 @@
1
+ # Method: writing the story
2
+
3
+ A story is written from the facts, for the family.
4
+
5
+ - Every statement rests on a recorded fact; say which ones.
6
+ - What is inferred is marked as inference; what is unknown is said to be
7
+ unknown. Never fill gaps with plausible fiction.
8
+ - Use the research language and plain words; explain old occupations and
9
+ terms the reader will not know.
10
+ - Write it to a file in notes/, then `strom story set P… --text @notes/story-P….md
11
+ --fact E… --fact E… --note "what is inferred"`: every fact it leans on goes
12
+ in --fact. A couple's story goes on the family (F…).
13
+ - It stays a draft; `--final` only when the user approved it.
@@ -0,0 +1,62 @@
1
+ # Method: reading scans
2
+
3
+ An image you open stays in your context and is paid for again on every turn.
4
+ Look at as few pixels as the question needs, and write down what you saw at once.
5
+
6
+ - **Look only through views**: `strom media view B0001:57` (record set and
7
+ image number) makes a file in `.strom/views/` — open that file. Whole images
8
+ come reduced: good for "is our surname on this page?", not for reading.
9
+ - **Find, then crop.** `--grid` overlays tenths with labels; read off where the
10
+ entry is and ask for exactly that part: `--crop 0.05,0.40,0.45,0.18` (x, y,
11
+ width, height as fractions) or `--half left|right` for one page of a spread.
12
+ A crop comes at full resolution (small ones enlarged). `--contrast` for faded
13
+ ink, `--rotate 90` for sideways pages.
14
+ - **Too small to read?** When a crop says it is enlarged, the scan has no more
15
+ detail there. A connector that can fetch a part of an image sharper says so
16
+ in that line: `strom fetch <connector> --recordset B0001 --images 57 --crop
17
+ …` (one request). Then view the same crop again: it comes from the sharper
18
+ part by itself.
19
+ - **Browsing a book is a reader's job**: `strom read B0001 --images 40-69
20
+ --question "…"` — readers in batches of ten, each with the full question;
21
+ their reports stay in notes/readings/ for the next session, you get the finds.
22
+ It waits for its readers (often ten minutes or more): run it in the
23
+ foreground and let it finish — your session ends with your turn, and
24
+ whatever is left running in the background ends with it.
25
+ (Your own subagents can read too, in batches of at most twelve — but what they
26
+ report is lost with your session unless you write it down.) Only the entries
27
+ that will be cited need your own eyes, at full resolution.
28
+ - Old handwriting is decoded, not copied: never a weaker model for handwriting,
29
+ never a guess. "Illegible" is a valid and valuable answer.
30
+ - **Extract everything the first time**: names, ages, house numbers,
31
+ occupations, godparents, witnesses, midwife, remarks in the margin.
32
+ - **Report image by image, as you go**: the image and page, what was found (or
33
+ nothing), what was illegible and where, the hand, how sure each name is.
34
+ - **Cite the image**: `strom source add … --media B0001:57 --locator "pag. 112,
35
+ 2nd entry"`; a searched range goes in `strom search add … --pages 40-69`.
36
+ - Page ↔ image: `strom recordset calibrate B0001 --point 57=112` (measured on
37
+ the image, never guessed); then `strom media view B0001 --page 112` works.
38
+ - **No images here yet:** a connector for that archive
39
+ (`strom connector list`) fetches them through strom:
40
+ `strom fetch <connector> <book> --images 40-69 --recordset B0001` — only the
41
+ images you need, never a whole book "just in case".
42
+ - **The archive has no connector yet: build one — now, you.** The user does
43
+ not know connectors exist and will not ask for one. Tell them in a sentence
44
+ ("for this archive I am preparing a downloader; it will fetch only the images
45
+ we need, slowly"), run `strom connector new <name> --url <portal>` and
46
+ follow its DISCOVERY.md (the portal's terms and robots.txt first, official
47
+ exports preferred), test it with `strom connector test`, then fetch. It is
48
+ built once per archive and serves every later task.
49
+ - You never download from an archive yourself (curl, a script, your browser
50
+ tools) — only through a connector, paced by strom. The user saves images by
51
+ hand only where the archive does not allow automation (the connector then
52
+ finds books and gives links only) or where a check stops it. Ask with
53
+ `strom task wait T… --images B0001:40-69 --on "…"`: strom makes the folder
54
+ of the inbox they go into, shows it to the user with the book's link, and
55
+ checks the numbers of what arrives. Write `--on` for the user, in their
56
+ language, so that they need nothing else: the book (title, call number),
57
+ its link, which images as the portal's viewer counts them (or the pages and
58
+ years, where you know only those), and that each is saved named by its
59
+ number (40.jpg). Where the portal gives only a small image, add: zoom in on
60
+ the entry and save that view too (40a.jpg); a full-resolution scan can be
61
+ ordered from the archive. Then take the next task; when the images are
62
+ registered the task comes back by itself.
@@ -0,0 +1,59 @@
1
+ # Method: recording an entry
2
+
3
+ Record a found entry in one batch: write the lines to a file in notes/, run
4
+ `strom batch --file notes/<file> --dry-run`, fix what it reports, then run it
5
+ without --dry-run. `#name` labels what a line creates, `@name` uses it later.
6
+
7
+ source add "Baptism of Jan Novák 1885" --kind baptism --recordset B0001 --media B0001:57 --locator "pag. 112, entry 2" --language la --information primary --transcript @notes/entry.txt #s
8
+ event add P0001 CHR --date "25 JUN 1885" --place "Týnec" --house 13 --cite @s --quote "baptizatus est" --with "godparent:Marie Dvořáková" --with "midwife:Anna Nová" --with "officiant:P. Josef Kříž" --status proven
9
+ cite E0001 @s --quote "natus 24. Junii" --status proven
10
+ name add P0002 "Marie /Svobodová/" --kind birth --cite @s --quote "Maria filia Josephi Svoboda"
11
+ person add "Josef /Svoboda/" --sex M --cite @s --information secondary #josef
12
+ family add --partner @josef --child P0002 --cite @s --information secondary
13
+ search add "Baptism of Jan Novák" --recordset B0001 --years 1884-1886 --pages 55-60 --method page-by-page --result found --found @s
14
+
15
+ - A fact already in the tree gets the citation (`cite E…`), not a second fact —
16
+ and what the record adds to it: `event edit E… --age husband:27 --age wife:17
17
+ --house 21 --with "witness:…" --with "officiant:…"` (filling in needs no reason).
18
+ - The names a record gives go on the person: `name add` (a maiden name completes
19
+ "Marie" to "Marie /Svobodová/"; a married name is `--kind married`).
20
+ - A record that names someone's parents (a grandchild's baptism naming the
21
+ grandparents) cites the family itself (`family add … --cite`, `cite F…`) and
22
+ the new people's names (`person add … --cite`) — with `--information
23
+ secondary`: the priest wrote down what he was told.
24
+ - What the entry says about each person it names becomes that person's fact,
25
+ dated by the entry: an occupation `event add P… OCCU --value "cottager"`, where
26
+ they live `event add P… RESI --place "Týnec" --house 7` — for the
27
+ grandparents too ("daughter of Jan Novák, cottager in Týnec No. 7"), cited
28
+ `--information secondary`. The next search identifies them by exactly these.
29
+ - Everyone the entry names in a role goes in `--with`: godparent, witness,
30
+ midwife, officiant, informant, other.
31
+ - A child the record does not name (stillborn, died unbaptised) has no given
32
+ name — never a description in its place: `person add "/Novák/" --sex M`, then
33
+ `event add P… DEAT --date … --age stillborn` (or a `--note`).
34
+ - Only what you read for sure goes into a field (`--date`, `--house`, `--age`,
35
+ `--cause`, a name). An uncertain word or digit is marked `[?]` in the
36
+ transcript and said in the fact's note ("house 21 or 27"); what a reader
37
+ reported as illegible stays illegible until you read it yourself. An empty
38
+ column is empty — nothing is filled in from what would be usual. A name
39
+ read only in part keeps its sure letters, `[?]` for the rest (`"Anna
40
+ /Kr[?]ková/"`), so the person can still be found; a date whose day is
41
+ unsure keeps the month (`--date "MAY 1850"`, note "12 or 17"). A
42
+ conclusion that stands on one hard cell (a house number, an age, a maiden
43
+ name) gets a blind second reading before it is `proven`: `strom read M…
44
+ --blind --crop x,y,w,h --question "Transcribe this entry"`.
45
+ - A place is the settlement only ("Týnec", not "Týnec No. 13"): the house
46
+ number goes in `--house 13`.
47
+ - What a finding means for other tasks: `strom task edit T… --note "…"`.
48
+ - Records about one person disagree (a name, a date, an age): record both
49
+ claims and your reasoning — `strom conflict add "<question>" --about P…
50
+ --claim "S…:<what it says>" --claim "S…:<what the other says>"`. Records
51
+ that give different parents are first two people (see core); they are one
52
+ person with a conflict only once something else proves it. What would decide
53
+ it, if not at hand, is a task about it: `strom task add … --about X…` (or H…).
54
+ `strom conflict resolve X… --resolution "…" --reasoning "…"` when the
55
+ evidence decides it. A decided
56
+ hypothesis that a new record overturns is decided again (the earlier
57
+ decision is kept): `strom hypothesis decide H… --decision "…" --reason "…"`.
58
+ People named wrongly by the weaker record keep that name as a variant
59
+ (`name add … --primary` for the right one); they are not deleted.
@@ -0,0 +1,10 @@
1
+ # Method: requests to archives
2
+
3
+ When records are not online, ask the holding archive.
4
+
5
+ - Write the request in the archive's language, short and precise: whose record,
6
+ which type, parish or office, years, what you already know (call number if
7
+ known). Offer to pay the research fee if the archive charges one.
8
+ - Save the text in `notes/` and tell the user — the user sends it, not you.
9
+ - Put the task on `strom task wait T… --on "reply from …"`. When the answer
10
+ comes, it becomes an input (`strom intake …`).
@@ -0,0 +1,17 @@
1
+ # Method: independent verification
2
+
3
+ A conclusion that stands on one hard-to-read cell (a house number, an age, a
4
+ first name) needs a second, independent reading.
5
+
6
+ - The second reader must be **blind**: do not tell them the expected value, the
7
+ name, the place or why it matters. Ask only what is written in that cell.
8
+ - One cell, one second reader — not the whole page again.
9
+ - Before asking, check whether two readings already exist (`strom find`).
10
+ - Certainty above "probable" only when two independent readings agree. When
11
+ they disagree, record a conflict with both readings.
12
+ - An independent reading of an entry you cited: `strom read M… --blind --crop
13
+ x,y,w,h --question "Transcribe this entry completely; mark what is uncertain
14
+ with [?]"` — a reader who knows nothing of your conclusion reads it at full
15
+ resolution. Where it differs from your reading (a digit, a name), look again
16
+ at that spot; what stays uncertain goes into the note, and a real
17
+ disagreement between two readings is a conflict, not a choice.
@@ -0,0 +1,23 @@
1
+ # strom plugins
2
+
3
+ Plugins extend strom on this computer: one folder per kind of plugin, and in
4
+ it one folder per plugin. The folder's name is the plugin's name.
5
+
6
+ plugins/
7
+ connectors/ downloaders for archive portals
8
+ README.md their interface (version 1)
9
+ example-archive/ one connector: connector.json and its program
10
+
11
+ **Install** a plugin by copying its folder in: `strom connector list` shows it
12
+ and it can run at once, paced by strom and only to the hosts it names. To be
13
+ asked before any plugin runs: `strom config set connectors.consent on`. A
14
+ plugin whose code goes round strom always needs your yes, in your terminal
15
+ (`strom allow connector <name>`).
16
+
17
+ **Build** one with your agent: `strom connector new <name> --url <portal>`.
18
+
19
+ **Remove** one by deleting its folder (or `strom connector remove <name>`).
20
+
21
+ Plugins are programs on this computer, put here by you: the `.gitignore` here
22
+ keeps them out of any repository. strom writes this file, the `.gitignore` and
23
+ `connectors/README.md`, and keeps them up to date.
@@ -0,0 +1,159 @@
1
+ # Building the connector for __TITLE__
2
+
3
+ You (the agent) build a connector for __URL__ together with the user. Work in
4
+ this folder only. The contract (interface 1) is `../README.md`: read it first.
5
+ strom runs the connector; you never run it yourself, you test it with
6
+ `strom connector test`. Talk to the user in their language.
7
+
8
+ ## 1. What does the portal allow? (before any code)
9
+
10
+ - **Terms of use** (Nutzungsbedingungen, podmínky užití, regulamin, conditions
11
+ d'utilisation, a reading-room rule …): find them, quote the sentences about
12
+ automated or bulk download, copying and reuse, and note the URL.
13
+ - **robots.txt** of each host: what is disallowed, and any crawl-delay.
14
+ - **Official ways to get the images**: a download of a whole volume (ZIP or
15
+ PDF), IIIF manifests, an API, a published list of books. Prefer them: they
16
+ are what the archive wants people to use.
17
+ - Write what you found into `connector.json` → `policy`:
18
+ - `terms` (the URL), `termsSummary` (one to three sentences, quoted where it
19
+ matters), `robots`, `officialExport`.
20
+ - `automation`, one of:
21
+ - `"allowed"`: the terms allow it or say nothing against it, and nothing
22
+ is got round.
23
+ - `"manual"`: the terms forbid automated or bulk download. The connector
24
+ only finds books and gives their links (`"can": ["find", "list"]`), and
25
+ the user downloads by hand into the inbox.
26
+ - `"unknown"`: you could not find out. Say so; the user decides.
27
+ - `pace`: slower than strom's default when the terms or robots.txt ask for
28
+ it, e.g. `{"minIntervalMs": 5000}` for a crawl-delay of 5.
29
+ - **Never get round a technical measure**: logins you do not have, captchas,
30
+ or tokens meant to stop scripts. Tiles are how many viewers show big images;
31
+ they are a measure against downloading only when the terms or the portal
32
+ say so — then stop and tell the user.
33
+ - **A bot check** (strom's probe says so: Imperva, Cloudflare — a page that
34
+ only runs a script): the portal answers a real browser only. Then the whole
35
+ connector goes through the user's browser: `"routes": ["browser"]` and
36
+ `"browser": {"pages": true}` (the contract, section 5), then
37
+ `strom connector use __NAME__ --via browser`. Browser tools come with your
38
+ next session; probes then go through the browser too. The user passes the
39
+ check in their own browser — never you. In the browser, look at the portal
40
+ as a person does, a page at a time; what you would script (a list of
41
+ addresses, a search to repeat) goes through `strom connector probe`, which
42
+ strom paces — not through fetch() in the tab.
43
+ - Tell the user what you found **before** you write code.
44
+
45
+ Read the portal with your web tools (pages, not images). Never download images
46
+ yourself. Once the hosts are in `connector.json`, read the portal through strom
47
+ as well: see step 2.
48
+
49
+ ## 2. Map the portal
50
+
51
+ - **Finding books** → `find`: how to find the books of a place (a catalogue
52
+ search, a parish tree, an API). A search form is fine: send it with `post()`.
53
+ - **One book** → `list` and `fetch`:
54
+ - how a book is identified (usually an ID in its URL);
55
+ - how many images it has;
56
+ - what the address of one image looks like.
57
+ - **The best address of an image**, in this order:
58
+ 1. the portal's own link to the full image, or a whole-volume download;
59
+ 2. a IIIF image: one request per image. Compare `info.json` (the original
60
+ size) with what `full/max` gives: some servers cap it.
61
+ Then the largest one request can give is the right choice. Do not ask
62
+ for hundreds of tiles.
63
+ 3. tiles (DeepZoom, Zoomify, IIIF tiles), only when nothing else exists and
64
+ the terms allow it. Save the tiles and give them to strom with
65
+ `imageFromTiles()`: strom puts them together. For `fetch` take the level
66
+ nearest 2000 px on the longer side (what a viewer shows of a page); for
67
+ `part` the tiles of that region at full size. Say in `README.md` how many
68
+ requests one image and one part cost.
69
+ Whatever the address, check what it gives: strom answers each saved image
70
+ with its `width` and `height`. A thumbnail or a stand-in (a tiny picture in
71
+ place of an image the portal will not give) is not a scan — some portals
72
+ answer so for some images only: find where the full image is then (another
73
+ address, tiles, a whole-volume download).
74
+ - **A part of an image, sharper** → `part`: when the whole image comes smaller
75
+ than the original, find whether one request can give a part of it at more
76
+ detail (IIIF: a region of the original, `x,y,w,h` or `pct:x,y,w,h`; a
77
+ viewer's zoomed-in link). Then add `"part"` to `can`: a reader asks for the
78
+ entry it cannot read, one request, instead of the whole image in tiles.
79
+ - **An account**: when the portal gives more to users who log in (full-size
80
+ images, a subscription), the connector may use the user's own account.
81
+ Never ask the user for it and never type it in: declare `login` in
82
+ `connector.json` (see the contract), and the user saves it in their own
83
+ terminal (`strom login __NAME__`). Map the login form without logging in.
84
+ - **The user's own browser** → `routes` and `locate`: when the portal gives
85
+ its images only to a browser (a login with a second factor, a session a
86
+ script cannot start), or the user wants them through theirs. Add
87
+ `"browser"` to `routes` (alone, when direct cannot work) and `"locate"` to
88
+ `can`, and answer `locate` with the addresses `fetch` would ask for,
89
+ downloading nothing. `browser.open`: the page where the user logs in, on the
90
+ images' site. You never log in in the browser: the user does.
91
+ - **Pages built by JavaScript** (a catalogue or a viewer that is an
92
+ application) show your web tools almost nothing. Read them through strom:
93
+ `strom connector probe __NAME__ <url>` makes one request, paced and within
94
+ `hosts`, and saves the answer into `.test/probe/`. Fetch the page, find the
95
+ script it loads (`<script src=…>`), probe that too, and search it for the
96
+ addresses the application calls (`/api/`, `.json`, image or file addresses):
97
+ `strom connector grep __NAME__ <text>` shows each hit with the text round it
98
+ (`--regex` for a pattern). Pages and scripts are often too big to read whole.
99
+ Then probe those addresses to see what they answer. A POST works too:
100
+ `--method POST --body '…' --header "Content-Type: application/json"`.
101
+ Probes keep the cookies servers set, like one visit in a browser: probe the
102
+ page that starts a session, then the search or the viewer (`--fresh` starts
103
+ a new visit). Each probe is one request: keep them few, a dozen or two for
104
+ a whole portal.
105
+ - **What the portal needs from a client**: a session (strom keeps its
106
+ cookies), a `Referer`, an `X-Requested-With` for its own scripts. Send
107
+ exactly what its pages send, nothing more.
108
+ - Write the mapping into `README.md`, with URLs and placeholders. Whoever
109
+ keeps the connector working later starts there.
110
+
111
+ ## 3. Write the connector
112
+
113
+ - Write `connector.ts` with the SDK (`sdk.ts`; do not change it):
114
+ - `get(url)` for pages and `post(url, {field: value})` for forms;
115
+ - `get(url, { save: "s0040.jpg", headers: { Referer: … } })` for images,
116
+ then `image(n, file, url)`;
117
+ - `book({...})` for what `find` and `list` found;
118
+ - for `part`: one `image(n, file, url)` for the part of image `n`;
119
+ - tiles only: `imageFromTiles(n, file, tiles, { width, height }, url)`;
120
+ - for `locate`: `located(n, src, url)` for each image — `src` the image's
121
+ own address, nothing fetched;
122
+ - with a login: `post(url, { name: login("user"), pass: login("password") })`,
123
+ only when `req.login` is true — strom puts the values in;
124
+ - `log()`, and `done()` at the end.
125
+ - **Never** use `fetch()`, http or net modules, child_process or curl. strom
126
+ checks the code and warns the user that it cannot pace or stop such
127
+ requests.
128
+ - Put in `hosts` of `connector.json` only the hosts the connector needs.
129
+ - Name the files by image number (`s0040.jpg`), fetch only the images asked
130
+ for, in order.
131
+
132
+ ## 4. Test with a few requests
133
+
134
+ strom connector test __NAME__ --find "<place>" --years 1780-1850
135
+ strom connector test __NAME__ --list <book>
136
+ strom connector test __NAME__ --fetch <book> --images 1-2
137
+ strom connector test __NAME__ --fetch <book> --images 2 --crop 0.5,0,0.5,0.5 (with part)
138
+ strom connector test __NAME__ --locate <book> --images 1-2 (with locate)
139
+
140
+ - Run each `strom` command on its own, not in a pipe or after `;`: your
141
+ permissions allow `strom …` and nothing around it.
142
+ - A test makes at most 10 requests (`--max 50` at most, for an image in
143
+ tiles). Its files go to `.test/` in this folder.
144
+ - strom shows each image's size (width × height). It should be the right page
145
+ and full size, not a thumbnail. Check that the number of images of a book
146
+ matches the portal.
147
+ - If strom answers with exit code 4 (consent required), stop and tell the
148
+ user: they give it in their own terminal after reading the warning
149
+ (`strom allow connector __NAME__`). You never do that.
150
+
151
+ ## 5. Hand over
152
+
153
+ Show the user:
154
+ - what the terms say (the policy);
155
+ - what the connector can do;
156
+ - the test results, with how many requests one image costs.
157
+
158
+ They decide whether to use it. From then on, `strom fetch __NAME__ …`
159
+ downloads through strom's limiter and registers what it fetched.