strom-research 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +373 -0
- package/README.md +142 -0
- package/assets/lang/cs.json +302 -0
- package/assets/lang/de.json +302 -0
- package/assets/method/core.md +43 -0
- package/assets/method/enrich.md +11 -0
- package/assets/method/intake.md +30 -0
- package/assets/method/link.md +28 -0
- package/assets/method/locate.md +28 -0
- package/assets/method/narrate.md +13 -0
- package/assets/method/reading.md +62 -0
- package/assets/method/recording.md +59 -0
- package/assets/method/request.md +10 -0
- package/assets/method/verify.md +17 -0
- package/assets/plugins/README.md +23 -0
- package/assets/plugins/connectors/DISCOVERY.md +159 -0
- package/assets/plugins/connectors/README.md +376 -0
- package/assets/plugins/connectors/sdk.ts +168 -0
- package/assets/plugins/connectors/template.ts +38 -0
- package/assets/plugins/gitignore +4 -0
- package/dist/agents/files.js +313 -0
- package/dist/agents/global.js +257 -0
- package/dist/agents/launch.js +36 -0
- package/dist/agents/profiles.js +95 -0
- package/dist/brief/brief.js +345 -0
- package/dist/cli/commit.js +44 -0
- package/dist/cli/context.js +311 -0
- package/dist/cli/execute.js +154 -0
- package/dist/cli/fixes.js +78 -0
- package/dist/cli/format.js +53 -0
- package/dist/cli/help.js +59 -0
- package/dist/cli/main.js +152 -0
- package/dist/cli/menu.js +212 -0
- package/dist/cli/registry.js +96 -0
- package/dist/cli/ui.js +266 -0
- package/dist/cli/wizard.js +142 -0
- package/dist/cli.js +14 -0
- package/dist/commands/analysis.js +622 -0
- package/dist/commands/batch.js +181 -0
- package/dist/commands/checks.js +153 -0
- package/dist/commands/connectors.js +1377 -0
- package/dist/commands/guide.js +160 -0
- package/dist/commands/index.js +19 -0
- package/dist/commands/intake.js +234 -0
- package/dist/commands/media.js +406 -0
- package/dist/commands/meta.js +195 -0
- package/dist/commands/output.js +117 -0
- package/dist/commands/people.js +664 -0
- package/dist/commands/read.js +199 -0
- package/dist/commands/research.js +139 -0
- package/dist/commands/session.js +605 -0
- package/dist/commands/setup.js +465 -0
- package/dist/commands/sources.js +634 -0
- package/dist/commands/start.js +383 -0
- package/dist/commands/story.js +75 -0
- package/dist/commands/tasks.js +436 -0
- package/dist/commands/trees.js +128 -0
- package/dist/core/actions.js +852 -0
- package/dist/core/age.js +95 -0
- package/dist/core/apps.js +74 -0
- package/dist/core/assets.js +34 -0
- package/dist/core/awake.js +33 -0
- package/dist/core/browser.js +281 -0
- package/dist/core/calibration.js +48 -0
- package/dist/core/check.js +112 -0
- package/dist/core/chromium.js +88 -0
- package/dist/core/config.js +348 -0
- package/dist/core/connector.js +811 -0
- package/dist/core/deps.js +73 -0
- package/dist/core/dialog.js +61 -0
- package/dist/core/errors.js +89 -0
- package/dist/core/evidence.js +58 -0
- package/dist/core/frontier.js +219 -0
- package/dist/core/gdate.js +77 -0
- package/dist/core/git.js +300 -0
- package/dist/core/guard.js +124 -0
- package/dist/core/http2.js +76 -0
- package/dist/core/import.js +541 -0
- package/dist/core/install.js +28 -0
- package/dist/core/integrity.js +219 -0
- package/dist/core/json.js +87 -0
- package/dist/core/lang.js +70 -0
- package/dist/core/live.js +244 -0
- package/dist/core/lock.js +112 -0
- package/dist/core/logins.js +67 -0
- package/dist/core/media.js +223 -0
- package/dist/core/model.js +101 -0
- package/dist/core/net.js +366 -0
- package/dist/core/open.js +29 -0
- package/dist/core/paths.js +84 -0
- package/dist/core/people.js +283 -0
- package/dist/core/phrases.js +85 -0
- package/dist/core/queue.js +113 -0
- package/dist/core/reader.js +76 -0
- package/dist/core/records.js +105 -0
- package/dist/core/roles.js +30 -0
- package/dist/core/schema.js +261 -0
- package/dist/core/seal.js +77 -0
- package/dist/core/self.js +40 -0
- package/dist/core/session.js +155 -0
- package/dist/core/shortcut.js +90 -0
- package/dist/core/stories.js +61 -0
- package/dist/core/stromapp.js +138 -0
- package/dist/core/text.js +104 -0
- package/dist/core/tree.js +507 -0
- package/dist/core/uninstall.js +128 -0
- package/dist/core/update.js +193 -0
- package/dist/core/validate.js +260 -0
- package/dist/core/views.js +164 -0
- package/dist/core/which.js +51 -0
- package/dist/core/workers.js +42 -0
- package/dist/gedcom/export.js +454 -0
- package/dist/gedcom/labels.js +103 -0
- package/dist/gedcom/lines.js +91 -0
- package/dist/gedcom/parse.js +53 -0
- package/dist/gedcom/validate.js +183 -0
- package/dist/image/image.js +223 -0
- package/dist/image/index.js +114 -0
- package/dist/image/jpeg-decode.js +552 -0
- package/dist/image/jpeg-encode.js +254 -0
- package/dist/image/png.js +241 -0
- package/dist/runners/antigravity.js +70 -0
- package/dist/runners/claude.js +179 -0
- package/dist/runners/codex.js +45 -0
- package/dist/runners/index.js +13 -0
- package/dist/runners/jsonl.js +86 -0
- package/dist/runners/opencode.js +50 -0
- package/dist/runners/runner.js +63 -0
- package/dist/runners/script.js +58 -0
- package/package.json +44 -0
|
@@ -0,0 +1,160 @@
|
|
|
1
|
+
// The agent guide: what an agent needs to know to do research with strom,
|
|
2
|
+
// with no other documentation. Keep it short — it is read at the start of
|
|
3
|
+
// sessions and costs tokens every time. Every command in it must exist
|
|
4
|
+
// (test/cli/examples.test.ts checks them against the registry).
|
|
5
|
+
import { langName } from "../core/lang.js";
|
|
6
|
+
export function guideText(lang) {
|
|
7
|
+
const language = lang
|
|
8
|
+
? `The research language of this tree is ${langName(lang)} (${lang}): talk to the user in ${langName(lang)} and write notes, tasks and stories in ${langName(lang)}. Keep transcripts of records in their original language.`
|
|
9
|
+
: "Talk to the user in their language. Set it as the research language: strom setup --lang <code> (cs, en, de, pl, …).";
|
|
10
|
+
return `STROM — genealogical research with an AI agent
|
|
11
|
+
|
|
12
|
+
WHAT THIS IS
|
|
13
|
+
strom is the research toolkit you (the agent) work through. It keeps the
|
|
14
|
+
evidence (people, families, events, sources, searches, tasks) in a git
|
|
15
|
+
repository, enforces genealogical method, and produces a GEDCOM file for
|
|
16
|
+
family tree programs such as the Strom app. You bring judgement and reading;
|
|
17
|
+
strom keeps the record straight.
|
|
18
|
+
|
|
19
|
+
HARD RULES
|
|
20
|
+
1. Never create, edit, delete or read files under data/ yourself. Every change
|
|
21
|
+
goes through a strom command. Direct edits are detected and block writing.
|
|
22
|
+
2. Never invent facts. What the user or a family tree says is a lead until a
|
|
23
|
+
record proves it.
|
|
24
|
+
3. Nothing is deleted. Wrong facts are retracted with --reason.
|
|
25
|
+
4. A namesake is not your person: check year, place, house and parents —
|
|
26
|
+
different parents mean two people, not a clerk's error.
|
|
27
|
+
5. Negative results are results: record every search, also when nothing
|
|
28
|
+
was found.
|
|
29
|
+
6. Do not answer consent questions yourself (exit code 4). Ask the user.
|
|
30
|
+
7. Never download from an archive yourself (curl, a script, browser tools).
|
|
31
|
+
Scans come through a connector (strom fetch), paced by strom — an archive
|
|
32
|
+
the research needs has none yet: build one first (strom connector new,
|
|
33
|
+
its DISCOVERY.md) and tell the user in a sentence. The user saves images
|
|
34
|
+
by hand only where the archive does not allow automation.
|
|
35
|
+
${language}
|
|
36
|
+
|
|
37
|
+
GETTING STARTED (once)
|
|
38
|
+
strom setup --yes --lang <code> the language the user speaks with you;
|
|
39
|
+
tell the user where the research lives
|
|
40
|
+
strom init "<family name>"
|
|
41
|
+
strom research new "<name>" --new-person "Josef /Novák/" --born "ABT 1885" --born-place "Kamenice nad Lipou"
|
|
42
|
+
strom intake --text "<what the user told you, in their words>"
|
|
43
|
+
strom intake <file or folder> documents, photos, a GEDCOM or Strom tree
|
|
44
|
+
|
|
45
|
+
EVERY SESSION
|
|
46
|
+
strom session start the brief: task, what is known, method
|
|
47
|
+
… work; record every finding at once …
|
|
48
|
+
strom task done T0001 --result "…" a complete negative search is a result
|
|
49
|
+
strom session close --summary "…" --next "…" (not finished: --continue)
|
|
50
|
+
in a conversation, the next task is best begun in a fresh context (Claude Code: /clear) —
|
|
51
|
+
suggest it to the user; many tasks waiting: the agent working on its own (strom run, the
|
|
52
|
+
user starts it from strom's menu), every task in a fresh session
|
|
53
|
+
|
|
54
|
+
RECORDING
|
|
55
|
+
A record (register entry, certificate) — the source first, then its facts:
|
|
56
|
+
strom source add "Křest Jana Nováka 1885" --kind baptism --recordset B0001 --locator "fol. 12, č. 3" --form original --information primary --transcript @zapis.txt
|
|
57
|
+
strom person add "Jan /Novák/" --sex M --born "24 JUN 1885" --born-place "Kamenice nad Lipou" --cite S0001
|
|
58
|
+
strom family add --partner P0002 --partner P0003 --child P0001 the baptism naming the parents proves the link
|
|
59
|
+
strom event add P0001 CHR --date "25 JUN 1885" --cite S0001 --locator "fol. 12" --with "godparent:Marie Dvořáková" --status proven
|
|
60
|
+
strom event add P0002 OCCU --value "rolník" --cite S0001
|
|
61
|
+
The same fact from a second record: strom cite E0001 S0002 (never a second BIRT).
|
|
62
|
+
Names a record gives: strom name add P0003 "Marie /Svobodová/" --kind birth --cite S0001
|
|
63
|
+
(a maiden name completes "Marie"; --kind married for a married name). A record
|
|
64
|
+
naming someone's parents (a grandchild's baptism) cites the family itself and
|
|
65
|
+
the new names: strom person add "Josef /Svoboda/" --cite S0001 --information secondary,
|
|
66
|
+
strom family add --partner P0004 --child P0003 --cite S0001 --information secondary.
|
|
67
|
+
Corrections — a person, a fact, a source:
|
|
68
|
+
strom person edit P0003 --sex F --name "…" --reason "…"
|
|
69
|
+
strom event edit E0001 --date "…" --reason "…"
|
|
70
|
+
strom source edit S0001 --translation "…"
|
|
71
|
+
One person recorded twice (an imported tree and the research) — parents
|
|
72
|
+
recorded twice are merged first:
|
|
73
|
+
strom family merge F0002 F0005 --reason "…"
|
|
74
|
+
strom person merge P0001 P0007 --reason "…"
|
|
75
|
+
Many facts from one record in one call — lines of commands, "#name" labels
|
|
76
|
+
what a line creates, "@name" uses it later:
|
|
77
|
+
strom batch 'person add "Anna /Svobodová/" --sex F #anna' 'family add --partner P0001 --partner @anna --married 1910 --cite S0001'
|
|
78
|
+
What the user remembers: strom intake --text "…", then people and facts
|
|
79
|
+
WITHOUT --cite — they stay leads. A document the user has but you have not
|
|
80
|
+
seen: record it as a source (--form original), cite it, and keep the facts
|
|
81
|
+
probable — proven needs your own reading of the record.
|
|
82
|
+
|
|
83
|
+
CERTAINTY
|
|
84
|
+
status of a fact = how sure the conclusion is:
|
|
85
|
+
proven you read the original record yourself and it states the fact
|
|
86
|
+
probable good evidence, but indirect, or a hard reading
|
|
87
|
+
possible weak evidence
|
|
88
|
+
lead family memory, a family tree, an index, a guess
|
|
89
|
+
evidence quality = what the record is (source --form original | derivative |
|
|
90
|
+
authored) and what it knew about THIS fact (--information primary: written at
|
|
91
|
+
the time by someone who knew | secondary). A baptism entry is primary for the
|
|
92
|
+
date of the baptism, secondary for the age of a parent: say so on the citation
|
|
93
|
+
(--information on cite/event add). A family tree, a memory or a compiled work
|
|
94
|
+
never makes a fact more than a lead; proven needs primary information.
|
|
95
|
+
An age gives a birth estimate: aged 25 at a marriage in 1910 →
|
|
96
|
+
strom event add P0001 BIRT --date "CAL 1885" --cite S0001 --quote "25 let" --information secondary
|
|
97
|
+
Ages as the record gives them: --age "25 let", "annorum 25", "3 months" (a
|
|
98
|
+
marriage gives both: strom family add … --age husband:25 --age wife:22).
|
|
99
|
+
|
|
100
|
+
SEARCHING
|
|
101
|
+
strom searched B0001 --years 1880-1890 before opening anything
|
|
102
|
+
strom search add "Křty Novák 1880-1890" --recordset B0001 --years 1880-1890 --surname Novák --method page-by-page --result negative
|
|
103
|
+
strom search edit Q0001 --recordset B0002 --reason "…" a search recorded wrongly is corrected, never added again
|
|
104
|
+
strom repo add … · strom recordset add … · strom place jurisdiction … where the records are
|
|
105
|
+
strom connector list connectors: they find books and fetch scans
|
|
106
|
+
strom fetch <connector> --find "<place>" --years 1780-1850
|
|
107
|
+
strom fetch <connector> <book> --images 40-69 --recordset B0001 registered at once
|
|
108
|
+
strom connector use <connector> --via browser only when the user asks: images through their own
|
|
109
|
+
browser (strom fetch then says what to do; --via direct: back)
|
|
110
|
+
strom connector new <name> --url <portal> an archive you need has none: build it (its DISCOVERY.md)
|
|
111
|
+
|
|
112
|
+
FINDING YOUR WAY
|
|
113
|
+
strom research show G0001 · strom person show P0001 · strom family show F0001
|
|
114
|
+
strom gaps · strom find <text> [<text>…] · strom frontier · strom task list
|
|
115
|
+
STORIES OF THE ANCESTORS (the setting stories — on by default; strom shows it)
|
|
116
|
+
Once records tell a person's life (a baptism and more facts from records), strom proposes a narrate
|
|
117
|
+
task: write the story for the family book — plain words, the research language, every statement on a
|
|
118
|
+
recorded fact (strom story set P… --text @notes/story-P….md --fact E… …). More facts later: it
|
|
119
|
+
proposes adding to it. A story is a draft until the user approves it (--final). Not told yet: when the
|
|
120
|
+
research starts, tell the user in a sentence, and that they may say no (strom config set stories no).
|
|
121
|
+
|
|
122
|
+
The result: strom export gedcom writes output/tree.ged (standard GEDCOM for any program) and
|
|
123
|
+
output/tree-strom.ged (for the Strom app) — both from the same evidence, also on session close.
|
|
124
|
+
Changes are committed automatically; you never run git yourself.
|
|
125
|
+
|
|
126
|
+
THE STROM APP — where the user sees the result
|
|
127
|
+
The Strom app (https://stromapp.info) is strom's companion: a free family tree app, no account,
|
|
128
|
+
the family's data stay on their computer; it shows the tree, the sources, a map, a family book.
|
|
129
|
+
When the user wants to see the results (or once, when the first ones are there), suggest it
|
|
130
|
+
gently, in a sentence or two — best installed as an app from the browser, then it works offline:
|
|
131
|
+
https://stromapp.info/run/ the app itself: opened in the browser, installed from there (the
|
|
132
|
+
install icon at the end of the address bar; Safari: File → Add to Dock)
|
|
133
|
+
strom app install opens it there and says where to click
|
|
134
|
+
strom app opens it — with this research when the app can take it (strom says so), and
|
|
135
|
+
run by you, the app follows the research live: what you record shows there
|
|
136
|
+
by itself — the best way to watch it grow; otherwise in the app: Import,
|
|
137
|
+
and output/tree-strom.ged
|
|
138
|
+
Without the app, the user can simply ask you about anyone in the tree, a family, a line, what is
|
|
139
|
+
proven and by which record: answer from strom (person show, family show, research show, find,
|
|
140
|
+
source show, story show, gaps, frontier) in plain words — names, dates, places, no IDs.
|
|
141
|
+
strom (no arguments) says whether it is installed here (results): not yet — offer to install it; installed —
|
|
142
|
+
just open it. A program they already use is fine too: output/tree.ged. They do not want it:
|
|
143
|
+
never again.
|
|
144
|
+
|
|
145
|
+
OUTPUT AND ERRORS
|
|
146
|
+
- One line per record, IDs first. Read the text; add --json only to parse.
|
|
147
|
+
- Listings are paged (--limit, --page). Do not pipe strom output.
|
|
148
|
+
- People can be named by ID (P0001) or name ("Jan Novák", no diacritics ok).
|
|
149
|
+
An ambiguous name lists candidates (exit 2) — then use the ID.
|
|
150
|
+
- Every error ends with "→ <what to run next>".
|
|
151
|
+
- Exit codes: 0 ok · 1 error · 2 usage/ambiguous · 3 needs input (ask the
|
|
152
|
+
user, then run the given command) · 4 needs consent (the USER must run the
|
|
153
|
+
given command in their terminal) · 5 locked by another session.
|
|
154
|
+
- Settings and how to override them: strom config where.
|
|
155
|
+
|
|
156
|
+
MORE
|
|
157
|
+
strom help <command> short help with examples
|
|
158
|
+
strom commands <group> --json options and examples of a group
|
|
159
|
+
`;
|
|
160
|
+
}
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
// Registers every command. Order here is the order in help listings.
|
|
2
|
+
import "./meta.js";
|
|
3
|
+
import "./start.js";
|
|
4
|
+
import "./trees.js";
|
|
5
|
+
import "./research.js";
|
|
6
|
+
import "./intake.js";
|
|
7
|
+
import "./tasks.js";
|
|
8
|
+
import "./people.js";
|
|
9
|
+
import "./batch.js";
|
|
10
|
+
import "./sources.js";
|
|
11
|
+
import "./media.js";
|
|
12
|
+
import "./connectors.js";
|
|
13
|
+
import "./read.js";
|
|
14
|
+
import "./analysis.js";
|
|
15
|
+
import "./story.js";
|
|
16
|
+
import "./output.js";
|
|
17
|
+
import "./session.js";
|
|
18
|
+
import "./checks.js";
|
|
19
|
+
import "./setup.js";
|
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
// intake · input list · show · skip · done — what a research starts from.
|
|
2
|
+
//
|
|
3
|
+
// Anything can be an input: a folder of scans and documents, a GEDCOM file,
|
|
4
|
+
// a tree exported from the Strom app, a text the user types. Inputs are the
|
|
5
|
+
// unchanged BASE of a research; the agent builds the working tree from them
|
|
6
|
+
// through tasks, and the tree it builds is exported as GEDCOM.
|
|
7
|
+
import fs from "node:fs";
|
|
8
|
+
import path from "node:path";
|
|
9
|
+
import { register } from "../cli/registry.js";
|
|
10
|
+
import { lines, moreLine, paginate, table, truncate } from "../cli/format.js";
|
|
11
|
+
import { UsageError } from "../core/errors.js";
|
|
12
|
+
import { likelyDuplicates } from "../core/people.js";
|
|
13
|
+
import { collectFiles, fileSha256, inputPath, MAX_IN_TREE, mimeOf, storeShared } from "../core/media.js";
|
|
14
|
+
import { create, normId, requireRecord, update } from "../core/records.js";
|
|
15
|
+
import { phrase } from "../core/phrases.js";
|
|
16
|
+
import { importGedcom, importStromJson, isStromJson } from "../core/import.js";
|
|
17
|
+
import { safeFolderName } from "../core/text.js";
|
|
18
|
+
function kindOf(file, text) {
|
|
19
|
+
const ext = path.extname(file).toLowerCase();
|
|
20
|
+
if (ext === ".ged")
|
|
21
|
+
return "tree";
|
|
22
|
+
if (ext === ".json" && text) {
|
|
23
|
+
try {
|
|
24
|
+
if (isStromJson(JSON.parse(text)))
|
|
25
|
+
return "tree";
|
|
26
|
+
}
|
|
27
|
+
catch {
|
|
28
|
+
// not JSON: other
|
|
29
|
+
}
|
|
30
|
+
}
|
|
31
|
+
if ([".jpg", ".jpeg", ".png", ".tif", ".tiff", ".gif", ".webp", ".heic", ".pdf"].includes(ext))
|
|
32
|
+
return "document";
|
|
33
|
+
if ([".txt", ".md", ".rtf", ".doc", ".docx", ".odt", ".html", ".csv"].includes(ext))
|
|
34
|
+
return "text";
|
|
35
|
+
return "other";
|
|
36
|
+
}
|
|
37
|
+
/** The research the inputs are for: the one named, else the only active one. */
|
|
38
|
+
function researchFor(tree, named) {
|
|
39
|
+
if (typeof named === "string")
|
|
40
|
+
return requireRecord(tree, named, "research").id;
|
|
41
|
+
const active = tree.list("research").filter((r) => r.state === "active");
|
|
42
|
+
return active.length === 1 ? active[0].id : undefined;
|
|
43
|
+
}
|
|
44
|
+
function intakeTask(tree, input, research, imported) {
|
|
45
|
+
return groupTask(tree, [input], research, imported);
|
|
46
|
+
}
|
|
47
|
+
/** One intake task for the inputs of one folder (or one file, or one text). */
|
|
48
|
+
function groupTask(tree, inputs, research, imported, folder) {
|
|
49
|
+
const input = inputs[0];
|
|
50
|
+
const many = inputs.length > 1;
|
|
51
|
+
const tree1 = input.kind === "tree" && !many;
|
|
52
|
+
const lang = tree.lang;
|
|
53
|
+
const what = many
|
|
54
|
+
? phrase(lang, "intake.files", { count: inputs.length, folder: folder ?? phrase(lang, "intake.the.material"), first: input.id, last: inputs.at(-1).id })
|
|
55
|
+
: tree1
|
|
56
|
+
? phrase(lang, imported ? "intake.tree.imported" : "intake.tree.read", { id: input.id, name: input.name })
|
|
57
|
+
: input.kind === "text"
|
|
58
|
+
? phrase(lang, input.text ? "intake.text" : "intake.text.file", { id: input.id, name: input.name })
|
|
59
|
+
: phrase(lang, "intake.other", { id: input.id, name: input.name });
|
|
60
|
+
const why = tree1
|
|
61
|
+
? imported
|
|
62
|
+
? phrase(lang, "intake.why.imported", { persons: imported.persons, matched: imported.matched ? phrase(lang, "intake.why.matched", { matched: imported.matched }) : "" })
|
|
63
|
+
: phrase(lang, "intake.why.read")
|
|
64
|
+
: phrase(lang, "intake.why");
|
|
65
|
+
const doneWhen = tree1 ? phrase(lang, imported ? "intake.done.imported" : "intake.done.read", { id: input.id }) : phrase(lang, "intake.done");
|
|
66
|
+
const ids = inputs.map((i) => i.id);
|
|
67
|
+
return create(tree, "task", { level: "intake", priority: 4, what, where: ids, why, doneWhen, subject: ids, research, state: "open", origin: tree.actor }, (id) => `+${id} task "${truncate(what, 60)}"`);
|
|
68
|
+
}
|
|
69
|
+
/** More images than this in one folder look like the scans of a book, not family documents. */
|
|
70
|
+
const BOOK_OF_SCANS = 20;
|
|
71
|
+
const IMAGE_EXT = new Set([".jpg", ".jpeg", ".png", ".tif", ".tiff", ".jp2", ".gif", ".webp", ".heic"]);
|
|
72
|
+
register({
|
|
73
|
+
path: ["intake"],
|
|
74
|
+
summary: "Take in material to research from: folders, files, a family tree (GEDCOM / Strom JSON), or text",
|
|
75
|
+
group: "inputs",
|
|
76
|
+
tree: true,
|
|
77
|
+
writes: true,
|
|
78
|
+
lock: "sections",
|
|
79
|
+
description: "Every file is registered unchanged as an input (duplicates skipped) and gets an intake task.\n" +
|
|
80
|
+
`Files up to ${MAX_IN_TREE / 1024 / 1024} MB are kept in the tree (and its history); bigger ones in shared/media.\n` +
|
|
81
|
+
"Family trees are imported at once — every person and fact as a lead citing that tree.",
|
|
82
|
+
args: [{ name: "paths", description: "files or folders", variadic: true }],
|
|
83
|
+
options: [
|
|
84
|
+
{ name: "text", type: "string", value: "<text>", description: "what the user tells you, as an input of its own" },
|
|
85
|
+
{ name: "no-import", type: "boolean", description: "register trees without importing them" },
|
|
86
|
+
{ name: "documents", type: "boolean", description: `a folder with more than ${BOOK_OF_SCANS} images really is family documents (not scans of a book)` },
|
|
87
|
+
{ name: "research", type: "string", value: "<G…>", description: "the research it is for (default: the only active one)" },
|
|
88
|
+
],
|
|
89
|
+
examples: ['strom intake ~/Downloads/rodina', "strom intake strom-export.json", 'strom intake --text "Děda Jan, narozen asi 1905 v Týnci, byl mlynář"', "strom intake ~/Downloads/rodina/dopis.txt --research G0001"],
|
|
90
|
+
run(ctx, { args, opts }) {
|
|
91
|
+
const tree = ctx.tree();
|
|
92
|
+
if (args.length === 0 && !opts.text)
|
|
93
|
+
throw new UsageError("give files/folders or --text");
|
|
94
|
+
const research = researchFor(tree, opts.research);
|
|
95
|
+
// Files per argument: a folder becomes one intake task, not one per file.
|
|
96
|
+
const groups = args.map((a) => {
|
|
97
|
+
const p = ctx.resolvePath(a);
|
|
98
|
+
if (!fs.existsSync(p))
|
|
99
|
+
throw new UsageError(`no such file or folder: ${a}`);
|
|
100
|
+
const folder = fs.statSync(p).isDirectory();
|
|
101
|
+
const files = collectFiles([p]);
|
|
102
|
+
const images = files.filter((f) => IMAGE_EXT.has(path.extname(f).toLowerCase())).length;
|
|
103
|
+
if (folder && images > BOOK_OF_SCANS && !opts.documents)
|
|
104
|
+
throw new UsageError(`${images} images in ${ctx.display(p)} look like the scans of a book, not family documents`, {
|
|
105
|
+
hint: `scans of a register: strom recordset add "<the book>" … then strom media add "${a}" --recordset B…\nif they are family documents after all: strom intake "${a}" --documents`,
|
|
106
|
+
});
|
|
107
|
+
return { folder: folder ? path.basename(p) : undefined, files };
|
|
108
|
+
});
|
|
109
|
+
const known = new Set(tree.list("input").map((i) => i.sha).filter(Boolean));
|
|
110
|
+
const report = [];
|
|
111
|
+
const added = [];
|
|
112
|
+
let skipped = 0;
|
|
113
|
+
tree.withTreeLock(() => {
|
|
114
|
+
for (const group of groups) {
|
|
115
|
+
const plain = [];
|
|
116
|
+
for (const file of group.files) {
|
|
117
|
+
const sha = fileSha256(file);
|
|
118
|
+
if (known.has(sha)) {
|
|
119
|
+
skipped++;
|
|
120
|
+
continue;
|
|
121
|
+
}
|
|
122
|
+
known.add(sha);
|
|
123
|
+
const size = fs.statSync(file).size;
|
|
124
|
+
const isText = size < 50 * 1024 * 1024 && [".ged", ".json", ".txt", ".md", ".csv"].includes(path.extname(file).toLowerCase());
|
|
125
|
+
const text = isText ? fs.readFileSync(file, "utf8") : undefined;
|
|
126
|
+
const kind = kindOf(file, text);
|
|
127
|
+
const id = tree.peekId("I");
|
|
128
|
+
let stored;
|
|
129
|
+
if (size <= MAX_IN_TREE) {
|
|
130
|
+
const rel = `inputs/${id}-${safeFolderName(path.basename(file))}`;
|
|
131
|
+
if (!tree.dryRun) {
|
|
132
|
+
tree.remember(path.join(tree.root, rel));
|
|
133
|
+
fs.copyFileSync(file, path.join(tree.root, rel));
|
|
134
|
+
}
|
|
135
|
+
stored = rel;
|
|
136
|
+
}
|
|
137
|
+
else
|
|
138
|
+
stored = `media:${path.relative(ctx.settings.shared().value, tree.dryRun ? file : storeShared(ctx.settings.shared().value, file, sha))}`;
|
|
139
|
+
const input = create(tree, "input", { name: path.basename(file), file: stored, sha, size, mime: mimeOf(file), from: ctx.display(file), kind, state: "new" }, (iid) => `+${iid} input ${kind} "${truncate(path.basename(file), 50)}"`);
|
|
140
|
+
let imported;
|
|
141
|
+
if (kind === "tree" && !opts["no-import"] && text !== undefined) {
|
|
142
|
+
imported = path.extname(file).toLowerCase() === ".ged"
|
|
143
|
+
? importGedcom(tree, text, { input: input.id, name: input.name, sha })
|
|
144
|
+
: importStromJson(tree, JSON.parse(text), { input: input.id, name: input.name });
|
|
145
|
+
update(tree, input.id, "input", (i) => ({ ...i, source: imported.source, imported: { persons: imported.persons, families: imported.families, sources: 1 } }), {
|
|
146
|
+
op: "input.imported",
|
|
147
|
+
summary: `${input.id} imported: ${imported.persons} persons, ${imported.families} families${imported.matched ? `, ${imported.matched} matched` : ""}`,
|
|
148
|
+
});
|
|
149
|
+
report.push(` ${input.id} tree: ${imported.persons} persons, ${imported.families} families, ${imported.events} facts as leads${imported.matched ? `, ${imported.matched} matched to existing` : ""}${imported.problems.length ? ` (${imported.problems.length} unreadable lines)` : ""}`);
|
|
150
|
+
if (imported.extended.length)
|
|
151
|
+
report.push(` added to what we have (leads — check them): ${imported.extended.slice(0, 6).join(" · ")}${imported.extended.length > 6 ? " …" : ""}`);
|
|
152
|
+
const system = `gedcom:${sha.slice(0, 12)}`;
|
|
153
|
+
const dupes = likelyDuplicates(tree, tree.list("person").filter((p) => p.refs?.some((r) => r.system === system)).map((p) => p.id));
|
|
154
|
+
if (dupes.length)
|
|
155
|
+
report.push(` probably already in the tree: ${dupes.map((d) => `${d.person.id}≈${d.same.id}`).join(", ")} (the review task lists them)`);
|
|
156
|
+
}
|
|
157
|
+
// A family tree gets its own review task; the other files of a folder share one.
|
|
158
|
+
if (kind === "tree" || !group.folder)
|
|
159
|
+
intakeTask(tree, input, research, imported);
|
|
160
|
+
else
|
|
161
|
+
plain.push(input);
|
|
162
|
+
added.push(input);
|
|
163
|
+
}
|
|
164
|
+
if (plain.length)
|
|
165
|
+
groupTask(tree, plain, research, undefined, group.folder);
|
|
166
|
+
}
|
|
167
|
+
if (typeof opts.text === "string" && opts.text.trim()) {
|
|
168
|
+
const input = create(tree, "input", { name: truncate(opts.text, 60), text: opts.text.trim(), kind: "text", state: "new", from: "user" }, (iid) => `+${iid} input text`);
|
|
169
|
+
intakeTask(tree, input, research);
|
|
170
|
+
added.push(input);
|
|
171
|
+
}
|
|
172
|
+
});
|
|
173
|
+
const text = lines(`${added.length} input(s) registered${skipped ? `, ${skipped} already known (same content)` : ""}${tree.dryRun ? " (dry run)" : ""}`, added.length ? table(added.map((i) => [` ${i.id}`, i.kind, truncate(i.name, 60)])) : undefined, ...report, tree.written.some((o) => o.op === "task.add") ? `\n${tree.written.filter((o) => o.op === "task.add").length} intake task(s) created → strom task next` : undefined);
|
|
174
|
+
return { text, data: { inputs: added, skipped } };
|
|
175
|
+
},
|
|
176
|
+
}, {
|
|
177
|
+
path: ["input", "list"],
|
|
178
|
+
summary: "Inputs of this tree and their state",
|
|
179
|
+
group: "inputs",
|
|
180
|
+
tree: true,
|
|
181
|
+
options: [{ name: "state", type: "string", value: "<state>", description: "new, processed or skipped" }, { name: "full", type: "boolean", description: "--json: whole records instead of one row each" }],
|
|
182
|
+
run(ctx, { opts }) {
|
|
183
|
+
const all = ctx.tree().list("input").filter((i) => !opts.state || i.state === opts.state);
|
|
184
|
+
const page = paginate(all, ctx.limit, ctx.page);
|
|
185
|
+
return {
|
|
186
|
+
text: all.length ? lines(table(page.items.map((i) => [i.id, i.kind, i.state, truncate(i.name, 60)])), moreLine(page, "strom input list")) : 'no inputs → strom intake <files or folder>',
|
|
187
|
+
data: { total: all.length, inputs: opts.full ? page.items : page.items.map((i) => ({ id: i.id, kind: i.kind, state: i.state, name: i.name })) },
|
|
188
|
+
};
|
|
189
|
+
},
|
|
190
|
+
}, {
|
|
191
|
+
path: ["input", "show"],
|
|
192
|
+
summary: "One input: where the file is (to read it), what came of it",
|
|
193
|
+
group: "inputs",
|
|
194
|
+
tree: true,
|
|
195
|
+
args: [{ name: "input", description: "input ID (I0001)", required: true }],
|
|
196
|
+
run(ctx, { args }) {
|
|
197
|
+
const tree = ctx.tree();
|
|
198
|
+
const i = requireRecord(tree, args[0], "input");
|
|
199
|
+
const abs = inputPath(tree, i);
|
|
200
|
+
const tasks = tree.list("task").filter((t) => t.where.includes(i.id));
|
|
201
|
+
const text = lines(`${i.id} ${i.name} [${i.kind} · ${i.state}]`, abs ? `file ${abs}` : undefined, i.from ? `from ${i.from}` : undefined, i.mime ? `type ${i.mime}${i.size !== undefined ? ` · ${Math.round(i.size / 1024)} kB` : ""}` : undefined, i.imported ? `import ${i.imported.persons} persons, ${i.imported.families} families as leads, cited as ${i.source}` : undefined, i.text ? `\n${i.text}` : undefined, !i.text && i.kind === "text" && abs && fs.existsSync(abs) && (i.size ?? 0) < 20_000 && /\.(txt|md|csv)$/i.test(abs) ? `\n${fs.readFileSync(abs, "utf8").trim()}` : undefined, tasks.length ? `\ntasks\n${table(tasks.map((t) => [` ${t.id}`, t.state, truncate(t.what, 60)]))}` : undefined, ...i.notes.map((n) => `note ${n.text}`), i.kind === "document" && abs
|
|
202
|
+
? `\nread the file (a big image: strom media view ${i.id} --grid, then --crop), then record what it says: strom source add … --input ${i.id}`
|
|
203
|
+
: undefined);
|
|
204
|
+
return { text, data: { input: i, path: abs, tasks: tasks.map((t) => t.id) } };
|
|
205
|
+
},
|
|
206
|
+
}, {
|
|
207
|
+
path: ["input", "done"],
|
|
208
|
+
summary: "Mark an input as processed",
|
|
209
|
+
group: "inputs",
|
|
210
|
+
tree: true,
|
|
211
|
+
writes: true,
|
|
212
|
+
args: [{ name: "input", description: "input ID", required: true }],
|
|
213
|
+
run(ctx, { args }) {
|
|
214
|
+
const tree = ctx.tree();
|
|
215
|
+
const id = normId(args[0], "input");
|
|
216
|
+
update(tree, id, "input", (i) => ({ ...i, state: "processed" }), { op: "input.done", summary: `${id} processed` });
|
|
217
|
+
return { text: lines(...tree.written.map((o) => o.summary)), data: { id } };
|
|
218
|
+
},
|
|
219
|
+
}, {
|
|
220
|
+
path: ["input", "skip"],
|
|
221
|
+
summary: "Mark an input as not useful (with the reason)",
|
|
222
|
+
group: "inputs",
|
|
223
|
+
tree: true,
|
|
224
|
+
writes: true,
|
|
225
|
+
args: [{ name: "input", description: "input ID", required: true }],
|
|
226
|
+
run(ctx, { args, opts }) {
|
|
227
|
+
const tree = ctx.tree();
|
|
228
|
+
if (!opts.reason)
|
|
229
|
+
throw new UsageError("--reason is required");
|
|
230
|
+
const id = normId(args[0], "input");
|
|
231
|
+
update(tree, id, "input", (i) => ({ ...i, state: "skipped" }), { op: "input.skip", summary: `${id} skipped`, reason: String(opts.reason) });
|
|
232
|
+
return { text: lines(...tree.written.map((o) => o.summary)), data: { id } };
|
|
233
|
+
},
|
|
234
|
+
});
|