@plantnet/planttaxomatcher 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +1305 -0
- package/package.json +42 -0
package/dist/index.js
ADDED
|
@@ -0,0 +1,1305 @@
|
|
|
1
|
+
#!/usr/bin/env node
|
|
2
|
+
|
|
3
|
+
// src/index.ts
|
|
4
|
+
import { promises as fs4 } from "fs";
|
|
5
|
+
import { parse as parsePath } from "path";
|
|
6
|
+
import { confirm, password } from "@inquirer/prompts";
|
|
7
|
+
import { Command } from "commander";
|
|
8
|
+
import kleur2 from "kleur";
|
|
9
|
+
|
|
10
|
+
// ../shared/src/schemas/identifiers.ts
|
|
11
|
+
import { z } from "zod";
|
|
12
|
+
var identifierNamespaceSchema = z.enum([
|
|
13
|
+
"wcvp:taxonID",
|
|
14
|
+
"wcvp:acceptedNameUsageID",
|
|
15
|
+
"ipni:lsid",
|
|
16
|
+
"gbif:taxonKey",
|
|
17
|
+
"powo:url"
|
|
18
|
+
]);
|
|
19
|
+
var taxonIdentifierSchema = z.object({
|
|
20
|
+
namespace: identifierNamespaceSchema,
|
|
21
|
+
value: z.string().min(1)
|
|
22
|
+
});
|
|
23
|
+
|
|
24
|
+
// ../shared/src/schemas/grading.ts
|
|
25
|
+
import { z as z2 } from "zod";
|
|
26
|
+
var matchStatusSchema = z2.enum(["matched", "ambiguous", "no_match", "error", "skipped"]);
|
|
27
|
+
var evidenceTypeSchema = z2.enum([
|
|
28
|
+
"local_exact",
|
|
29
|
+
"local_canonical_unique",
|
|
30
|
+
"external_validated",
|
|
31
|
+
"local_fuzzy",
|
|
32
|
+
"external_fuzzy",
|
|
33
|
+
"team_history",
|
|
34
|
+
"llm"
|
|
35
|
+
]);
|
|
36
|
+
var gradeSchema = z2.enum(["A", "B", "C"]);
|
|
37
|
+
var reviewStatusSchema = z2.enum(["not_required", "pending", "accepted", "rejected", "overridden"]);
|
|
38
|
+
var identificationQualifierSchema = z2.enum([
|
|
39
|
+
"none",
|
|
40
|
+
"cf",
|
|
41
|
+
"aff",
|
|
42
|
+
"sensu_lato",
|
|
43
|
+
"sensu_stricto",
|
|
44
|
+
"aggregate",
|
|
45
|
+
"complex",
|
|
46
|
+
"sp",
|
|
47
|
+
"spp",
|
|
48
|
+
"indet",
|
|
49
|
+
"cultivar",
|
|
50
|
+
"hybrid_formula",
|
|
51
|
+
"informal"
|
|
52
|
+
]);
|
|
53
|
+
var scoreBreakdownSchema = z2.object({
|
|
54
|
+
canonicalSimilarity: z2.number().min(0).max(1).optional(),
|
|
55
|
+
authorSimilarity: z2.number().min(0).max(1).optional(),
|
|
56
|
+
rankCompatibility: z2.number().min(0).max(1).optional(),
|
|
57
|
+
familyCompatibility: z2.number().min(0).max(1).optional(),
|
|
58
|
+
providerAgreement: z2.number().min(0).max(1).optional()
|
|
59
|
+
});
|
|
60
|
+
|
|
61
|
+
// ../shared/src/schemas/job.ts
|
|
62
|
+
import { z as z3 } from "zod";
|
|
63
|
+
var jobStatusSchema = z3.enum([
|
|
64
|
+
"queued",
|
|
65
|
+
"parsing",
|
|
66
|
+
"matching",
|
|
67
|
+
"paused",
|
|
68
|
+
"cancelled",
|
|
69
|
+
"completed",
|
|
70
|
+
"failed"
|
|
71
|
+
]);
|
|
72
|
+
var authorModeSchema = z3.enum(["ignore", "prefer", "strict"]);
|
|
73
|
+
var reviewModeSchema = z3.enum(["off", "recommended", "strict"]);
|
|
74
|
+
var jobConfigSchema = z3.object({
|
|
75
|
+
nameColumn: z3.string().min(1),
|
|
76
|
+
idColumn: z3.string().nullable().optional(),
|
|
77
|
+
familyColumn: z3.string().nullable().optional(),
|
|
78
|
+
genusColumn: z3.string().nullable().optional(),
|
|
79
|
+
rankColumn: z3.string().nullable().optional(),
|
|
80
|
+
authorColumn: z3.string().nullable().optional(),
|
|
81
|
+
sourceReferentialColumn: z3.string().nullable().optional(),
|
|
82
|
+
sourceIdColumn: z3.string().nullable().optional(),
|
|
83
|
+
authorMode: authorModeSchema.default("prefer"),
|
|
84
|
+
matchAuthors: z3.boolean().default(true),
|
|
85
|
+
parallelism: z3.number().int().min(1).max(10).default(4),
|
|
86
|
+
allowFuzzy: z3.boolean().default(true),
|
|
87
|
+
allowLlm: z3.boolean().default(false),
|
|
88
|
+
llmCostCapCents: z3.number().int().min(0).default(500),
|
|
89
|
+
reviewMode: reviewModeSchema.default("recommended"),
|
|
90
|
+
exportConfirmedOnly: z3.boolean().default(false),
|
|
91
|
+
/**
|
|
92
|
+
* Keep the identified accepted name at species level: when an input
|
|
93
|
+
* resolves to an infraspecific accepted taxon (Variety / Subspecies /
|
|
94
|
+
* Form / …), collapse it up to its parent Species. Default true; set false
|
|
95
|
+
* to keep the exact infraspecific accepted taxon.
|
|
96
|
+
*/
|
|
97
|
+
speciesLevelAcceptedOnly: z3.boolean().default(true),
|
|
98
|
+
/**
|
|
99
|
+
* Which WCVP snapshot to match against. When omitted, the API picks the
|
|
100
|
+
* most-recently-imported snapshot at submit time. The job stores the
|
|
101
|
+
* resolved version on `jobs.referential_version` so re-runs are
|
|
102
|
+
* reproducible even if a newer snapshot lands later.
|
|
103
|
+
*/
|
|
104
|
+
referentialVersion: z3.string().min(1).optional()
|
|
105
|
+
});
|
|
106
|
+
var wcvpSnapshotSummarySchema = z3.object({
|
|
107
|
+
version: z3.string(),
|
|
108
|
+
recordCount: z3.number().int().nonnegative(),
|
|
109
|
+
importedAt: z3.string().datetime(),
|
|
110
|
+
isLatest: z3.boolean()
|
|
111
|
+
});
|
|
112
|
+
var idTypeDetectionSchema = z3.object({
|
|
113
|
+
sampleSize: z3.number().int().nonnegative(),
|
|
114
|
+
counts: z3.record(z3.string(), z3.number().int().nonnegative()),
|
|
115
|
+
dominant: z3.string().nullable(),
|
|
116
|
+
dominantConfidence: z3.number().min(0).max(1),
|
|
117
|
+
minorityExamples: z3.array(
|
|
118
|
+
z3.object({
|
|
119
|
+
rowIndex: z3.number().int().nonnegative(),
|
|
120
|
+
value: z3.string(),
|
|
121
|
+
type: z3.string()
|
|
122
|
+
})
|
|
123
|
+
)
|
|
124
|
+
});
|
|
125
|
+
var publicAccessSchema = z3.enum(["none", "read", "review"]);
|
|
126
|
+
var jobSummarySchema = z3.object({
|
|
127
|
+
id: z3.string().ulid(),
|
|
128
|
+
teamId: z3.string().uuid(),
|
|
129
|
+
userId: z3.string().uuid(),
|
|
130
|
+
/** Optional user-supplied label. URL still uses `id`; null when unset. */
|
|
131
|
+
name: z3.string().nullable(),
|
|
132
|
+
/** Display name of the user who submitted the job (from their token's
|
|
133
|
+
* user). Null when unavailable (e.g. mutation responses). */
|
|
134
|
+
submittedBy: z3.string().nullable(),
|
|
135
|
+
status: jobStatusSchema,
|
|
136
|
+
/** 'none' = private. 'read'/'review' = anyone with the link, no sign-in. */
|
|
137
|
+
publicAccess: publicAccessSchema,
|
|
138
|
+
referentialVersion: z3.string(),
|
|
139
|
+
idColumn: z3.string().nullable(),
|
|
140
|
+
idTypeDetection: idTypeDetectionSchema.nullable(),
|
|
141
|
+
totalRows: z3.number().int().nonnegative(),
|
|
142
|
+
uniqueQueries: z3.number().int().nonnegative(),
|
|
143
|
+
processedRows: z3.number().int().nonnegative(),
|
|
144
|
+
matchedRows: z3.number().int().nonnegative(),
|
|
145
|
+
ambiguousRows: z3.number().int().nonnegative(),
|
|
146
|
+
errorRows: z3.number().int().nonnegative(),
|
|
147
|
+
needsReviewRows: z3.number().int().nonnegative(),
|
|
148
|
+
llmCostCents: z3.number().int().nonnegative(),
|
|
149
|
+
llmCostCapCents: z3.number().int().nonnegative(),
|
|
150
|
+
createdAt: z3.string().datetime(),
|
|
151
|
+
startedAt: z3.string().datetime().nullable(),
|
|
152
|
+
completedAt: z3.string().datetime().nullable(),
|
|
153
|
+
expiresAt: z3.string().datetime()
|
|
154
|
+
});
|
|
155
|
+
var jobPublicAccessUpdateSchema = z3.object({
|
|
156
|
+
publicAccess: publicAccessSchema
|
|
157
|
+
});
|
|
158
|
+
var progressEventSchema = z3.object({
|
|
159
|
+
type: z3.enum(["progress", "status", "error", "completed"]),
|
|
160
|
+
jobId: z3.string().ulid(),
|
|
161
|
+
timestamp: z3.string().datetime(),
|
|
162
|
+
processedRows: z3.number().int().nonnegative().optional(),
|
|
163
|
+
totalRows: z3.number().int().nonnegative().optional(),
|
|
164
|
+
status: jobStatusSchema.optional(),
|
|
165
|
+
message: z3.string().optional()
|
|
166
|
+
});
|
|
167
|
+
var taxonIdentifierRefSchema = z3.object({
|
|
168
|
+
namespace: z3.string(),
|
|
169
|
+
value: z3.string()
|
|
170
|
+
});
|
|
171
|
+
var jobRowSummarySchema = z3.object({
|
|
172
|
+
id: z3.string().uuid(),
|
|
173
|
+
rowIndex: z3.number().int().nonnegative(),
|
|
174
|
+
inputName: z3.string().nullable(),
|
|
175
|
+
inputId: z3.string().nullable(),
|
|
176
|
+
inputFamily: z3.string().nullable(),
|
|
177
|
+
matchStatus: matchStatusSchema.nullable(),
|
|
178
|
+
grade: gradeSchema.nullable(),
|
|
179
|
+
evidenceType: z3.string().nullable(),
|
|
180
|
+
reviewStatus: reviewStatusSchema,
|
|
181
|
+
matchQueryId: z3.string().uuid().nullable(),
|
|
182
|
+
confidence: z3.number().nullable(),
|
|
183
|
+
layer: z3.string().nullable(),
|
|
184
|
+
flags: z3.array(z3.string()).nullable(),
|
|
185
|
+
candidateAcceptedName: z3.string().nullable(),
|
|
186
|
+
/** Authorship of the accepted taxon (distinct from `candidateAuthorship`
|
|
187
|
+
* which carries the matched-row author when the match resolves via a synonym). */
|
|
188
|
+
candidateAcceptedAuthorship: z3.string().nullable(),
|
|
189
|
+
candidateAcceptedIdentifiers: z3.array(taxonIdentifierRefSchema).nullable(),
|
|
190
|
+
candidateScientificName: z3.string().nullable(),
|
|
191
|
+
candidateAuthorship: z3.string().nullable(),
|
|
192
|
+
candidateFamily: z3.string().nullable(),
|
|
193
|
+
candidateReason: z3.string().nullable(),
|
|
194
|
+
candidateTargetUrl: z3.string().nullable()
|
|
195
|
+
});
|
|
196
|
+
var gradeCountsSchema = z3.object({
|
|
197
|
+
A: z3.number().int().nonnegative(),
|
|
198
|
+
B: z3.number().int().nonnegative(),
|
|
199
|
+
C: z3.number().int().nonnegative(),
|
|
200
|
+
ungraded: z3.number().int().nonnegative()
|
|
201
|
+
});
|
|
202
|
+
var statusCountsSchema = z3.object({
|
|
203
|
+
matched: z3.number().int().nonnegative(),
|
|
204
|
+
ambiguous: z3.number().int().nonnegative(),
|
|
205
|
+
no_match: z3.number().int().nonnegative(),
|
|
206
|
+
error: z3.number().int().nonnegative(),
|
|
207
|
+
skipped: z3.number().int().nonnegative(),
|
|
208
|
+
/** Rows whose match query exists but hasn't been processed yet (mid-run).
|
|
209
|
+
* Distinct from `skipped` (no query — empty/guarded input). Not part of the
|
|
210
|
+
* outcome funnel; the SPA can show it as in-progress. */
|
|
211
|
+
pending: z3.number().int().nonnegative()
|
|
212
|
+
});
|
|
213
|
+
var jobRowsPageSchema = z3.object({
|
|
214
|
+
rows: z3.array(jobRowSummarySchema),
|
|
215
|
+
/** Row count matching the CURRENT filter (not the whole job) — drives
|
|
216
|
+
* pagination so page count adapts to active grade/status/search/conf filters. */
|
|
217
|
+
total: z3.number().int().nonnegative(),
|
|
218
|
+
offset: z3.number().int().nonnegative(),
|
|
219
|
+
limit: z3.number().int().positive(),
|
|
220
|
+
/** Per-grade row counts for the whole job, independent of the current
|
|
221
|
+
* filter. Used by the SPA to show "B (12)" next to each grade chip. */
|
|
222
|
+
gradeCounts: gradeCountsSchema,
|
|
223
|
+
/** Per-status row counts (matched / ambiguous / no_match / error /
|
|
224
|
+
* skipped), same scope + purpose as `gradeCounts`. */
|
|
225
|
+
statusCounts: statusCountsSchema
|
|
226
|
+
});
|
|
227
|
+
var wcvpSearchHitSchema = z3.object({
|
|
228
|
+
taxonId: z3.string(),
|
|
229
|
+
scientificName: z3.string(),
|
|
230
|
+
canonicalName: z3.string(),
|
|
231
|
+
authorship: z3.string().nullable(),
|
|
232
|
+
rank: z3.string().nullable(),
|
|
233
|
+
taxonomicStatus: z3.string().nullable(),
|
|
234
|
+
family: z3.string().nullable(),
|
|
235
|
+
acceptedTaxonId: z3.string().nullable(),
|
|
236
|
+
acceptedName: z3.string().nullable(),
|
|
237
|
+
similarity: z3.number()
|
|
238
|
+
});
|
|
239
|
+
var wcvpSearchResponseSchema = z3.array(wcvpSearchHitSchema);
|
|
240
|
+
var jobDiffRowSchema = z3.object({
|
|
241
|
+
rowId: z3.string().uuid(),
|
|
242
|
+
rowIndex: z3.number().int().nonnegative(),
|
|
243
|
+
inputName: z3.string().nullable(),
|
|
244
|
+
inputFamily: z3.string().nullable(),
|
|
245
|
+
matchedName: z3.string().nullable(),
|
|
246
|
+
matchedAuthorship: z3.string().nullable(),
|
|
247
|
+
matchedTaxonomicStatus: z3.string().nullable(),
|
|
248
|
+
acceptedName: z3.string().nullable(),
|
|
249
|
+
acceptedAuthorship: z3.string().nullable(),
|
|
250
|
+
acceptedFamily: z3.string().nullable(),
|
|
251
|
+
isSynonym: z3.boolean(),
|
|
252
|
+
familyChanged: z3.boolean(),
|
|
253
|
+
grade: gradeSchema.nullable(),
|
|
254
|
+
reviewStatus: reviewStatusSchema,
|
|
255
|
+
targetUrl: z3.string().nullable()
|
|
256
|
+
});
|
|
257
|
+
var jobDiffPageSchema = z3.object({
|
|
258
|
+
rows: z3.array(jobDiffRowSchema),
|
|
259
|
+
total: z3.number().int().nonnegative()
|
|
260
|
+
});
|
|
261
|
+
var reviewActionSchema = z3.enum(["accept", "reject"]);
|
|
262
|
+
var reviewRequestSchema = z3.object({
|
|
263
|
+
action: reviewActionSchema,
|
|
264
|
+
candidateId: z3.string().uuid()
|
|
265
|
+
});
|
|
266
|
+
var bulkReviewRequestSchema = z3.object({
|
|
267
|
+
action: reviewActionSchema,
|
|
268
|
+
filter: z3.object({
|
|
269
|
+
grade: gradeSchema.optional(),
|
|
270
|
+
matchStatus: matchStatusSchema.optional(),
|
|
271
|
+
family: z3.string().optional()
|
|
272
|
+
}).default({}),
|
|
273
|
+
dryRun: z3.boolean().default(false)
|
|
274
|
+
});
|
|
275
|
+
var bulkReviewResponseSchema = z3.object({
|
|
276
|
+
action: reviewActionSchema,
|
|
277
|
+
matchedQueries: z3.number().int().nonnegative(),
|
|
278
|
+
dryRun: z3.boolean(),
|
|
279
|
+
appliedAt: z3.string().datetime().nullable()
|
|
280
|
+
});
|
|
281
|
+
var reviewResponseSchema = z3.object({
|
|
282
|
+
matchQueryId: z3.string().uuid(),
|
|
283
|
+
action: reviewActionSchema,
|
|
284
|
+
candidateId: z3.string().uuid(),
|
|
285
|
+
reviewStatus: reviewStatusSchema,
|
|
286
|
+
reviewEventId: z3.string().uuid()
|
|
287
|
+
});
|
|
288
|
+
var matchAttemptSummarySchema = z3.object({
|
|
289
|
+
id: z3.string().uuid(),
|
|
290
|
+
layer: z3.string(),
|
|
291
|
+
provider: z3.string(),
|
|
292
|
+
status: z3.string(),
|
|
293
|
+
durationMs: z3.number().int().nonnegative(),
|
|
294
|
+
// JSONB fields accept any shape. Note: `z.unknown()` infers as optional,
|
|
295
|
+
// so consumers should guard for `undefined` even though we always send
|
|
296
|
+
// them as part of the response.
|
|
297
|
+
query: z3.unknown(),
|
|
298
|
+
rawResponseSummary: z3.unknown().nullable(),
|
|
299
|
+
createdAt: z3.string().datetime()
|
|
300
|
+
});
|
|
301
|
+
var matchCandidateSummarySchema = z3.object({
|
|
302
|
+
id: z3.string().uuid(),
|
|
303
|
+
attemptId: z3.string().uuid(),
|
|
304
|
+
source: z3.string(),
|
|
305
|
+
sourceId: z3.string().nullable(),
|
|
306
|
+
scientificName: z3.string(),
|
|
307
|
+
canonicalName: z3.string(),
|
|
308
|
+
authorship: z3.string().nullable(),
|
|
309
|
+
rank: z3.string().nullable(),
|
|
310
|
+
taxonomicStatus: z3.string().nullable(),
|
|
311
|
+
acceptedName: z3.string().nullable(),
|
|
312
|
+
acceptedAuthorship: z3.string().nullable(),
|
|
313
|
+
acceptedIdentifiers: z3.array(taxonIdentifierRefSchema).nullable(),
|
|
314
|
+
identifiers: z3.array(taxonIdentifierRefSchema),
|
|
315
|
+
family: z3.string().nullable(),
|
|
316
|
+
confidence: z3.number().nullable(),
|
|
317
|
+
grade: gradeSchema.nullable(),
|
|
318
|
+
reason: z3.string().nullable(),
|
|
319
|
+
flags: z3.array(z3.string()).nullable(),
|
|
320
|
+
state: z3.enum(["proposed", "accepted", "rejected", "overridden"]),
|
|
321
|
+
targetUrl: z3.string().nullable()
|
|
322
|
+
});
|
|
323
|
+
var matchQueryDetailSchema = z3.object({
|
|
324
|
+
id: z3.string().uuid(),
|
|
325
|
+
jobId: z3.string().ulid(),
|
|
326
|
+
normalizedInput: z3.string(),
|
|
327
|
+
parsed: z3.unknown(),
|
|
328
|
+
matchStatus: matchStatusSchema.nullable(),
|
|
329
|
+
evidenceType: z3.string().nullable(),
|
|
330
|
+
grade: gradeSchema.nullable(),
|
|
331
|
+
confidence: z3.number().nullable(),
|
|
332
|
+
flags: z3.array(z3.string()).nullable(),
|
|
333
|
+
reviewStatus: reviewStatusSchema,
|
|
334
|
+
selectedCandidateId: z3.string().uuid().nullable()
|
|
335
|
+
});
|
|
336
|
+
var queryCandidatesResponseSchema = z3.object({
|
|
337
|
+
query: matchQueryDetailSchema,
|
|
338
|
+
candidates: z3.array(matchCandidateSummarySchema),
|
|
339
|
+
attempts: z3.array(matchAttemptSummarySchema)
|
|
340
|
+
});
|
|
341
|
+
var reviewChatMessageSchema = z3.object({
|
|
342
|
+
role: z3.enum(["user", "assistant"]),
|
|
343
|
+
content: z3.string().min(1).max(4e3)
|
|
344
|
+
});
|
|
345
|
+
var reviewChatBodySchema = z3.object({
|
|
346
|
+
messages: z3.array(reviewChatMessageSchema).min(1).max(40)
|
|
347
|
+
});
|
|
348
|
+
var reviewChatResponseSchema = z3.object({
|
|
349
|
+
reply: z3.string(),
|
|
350
|
+
model: z3.string()
|
|
351
|
+
});
|
|
352
|
+
var replayableLayerSchema = z3.enum(["L1", "L2", "L3", "L4", "L5", "L6", "L7"]);
|
|
353
|
+
var replayLayerBodySchema = z3.object({ layer: replayableLayerSchema });
|
|
354
|
+
var replayCandidateSchema = z3.object({
|
|
355
|
+
scientificName: z3.string(),
|
|
356
|
+
canonicalName: z3.string(),
|
|
357
|
+
authorship: z3.string().nullable(),
|
|
358
|
+
rank: z3.string().nullable(),
|
|
359
|
+
taxonomicStatus: z3.string().nullable(),
|
|
360
|
+
acceptedName: z3.string().nullable(),
|
|
361
|
+
acceptedAuthorship: z3.string().nullable(),
|
|
362
|
+
family: z3.string().nullable(),
|
|
363
|
+
/** WCVP taxon id of the matched row (null for unresolved external proposals). */
|
|
364
|
+
taxonId: z3.string().nullable(),
|
|
365
|
+
/** WCVP taxon id of the resolved accepted taxon (null when unresolved). */
|
|
366
|
+
acceptedTaxonId: z3.string().nullable(),
|
|
367
|
+
confidence: z3.number().nullable(),
|
|
368
|
+
reason: z3.string().nullable(),
|
|
369
|
+
flags: z3.array(z3.string()),
|
|
370
|
+
targetUrl: z3.string().nullable(),
|
|
371
|
+
/** Provider that proposed this candidate (e.g. tnrs, gbif, openrouter,
|
|
372
|
+
* wcvp-local). */
|
|
373
|
+
provider: z3.string().nullable()
|
|
374
|
+
});
|
|
375
|
+
var replayAttemptSchema = z3.object({
|
|
376
|
+
provider: z3.string(),
|
|
377
|
+
status: z3.string(),
|
|
378
|
+
durationMs: z3.number().int().nonnegative(),
|
|
379
|
+
summary: z3.unknown().nullable()
|
|
380
|
+
});
|
|
381
|
+
var replayLayerResultSchema = z3.object({
|
|
382
|
+
layer: replayableLayerSchema,
|
|
383
|
+
kind: z3.enum(["unique", "ambiguous", "miss"]),
|
|
384
|
+
reason: z3.string().nullable(),
|
|
385
|
+
/** Human-facing note when the layer couldn't run as-is (e.g. LLM not
|
|
386
|
+
* configured, no external providers enabled). */
|
|
387
|
+
note: z3.string().nullable(),
|
|
388
|
+
providers: z3.array(z3.string()),
|
|
389
|
+
durationMs: z3.number().int().nonnegative(),
|
|
390
|
+
/** What was actually submitted to the layer (post gnparser). */
|
|
391
|
+
query: z3.object({
|
|
392
|
+
canonical: z3.string(),
|
|
393
|
+
authorship: z3.string().nullable(),
|
|
394
|
+
inputId: z3.string().nullable()
|
|
395
|
+
}),
|
|
396
|
+
candidates: z3.array(replayCandidateSchema),
|
|
397
|
+
attempts: z3.array(replayAttemptSchema)
|
|
398
|
+
});
|
|
399
|
+
|
|
400
|
+
// ../shared/src/schemas/auth.ts
|
|
401
|
+
import { z as z4 } from "zod";
|
|
402
|
+
var tokenScopeSchema = z4.enum([
|
|
403
|
+
"submit:job",
|
|
404
|
+
"read:job",
|
|
405
|
+
"cancel:job",
|
|
406
|
+
"review:job",
|
|
407
|
+
"override:match",
|
|
408
|
+
"download:job",
|
|
409
|
+
"admin:tokens",
|
|
410
|
+
"admin:jobs",
|
|
411
|
+
"admin:teams"
|
|
412
|
+
]);
|
|
413
|
+
var ALL_TOKEN_SCOPES = tokenScopeSchema.options;
|
|
414
|
+
var tokenStringSchema = z4.string().regex(/^ptm_[a-zA-Z0-9]{12}_[a-zA-Z0-9]{32}$/, "Malformed token");
|
|
415
|
+
var meSchema = z4.object({
|
|
416
|
+
userId: z4.string().uuid(),
|
|
417
|
+
teamId: z4.string().uuid(),
|
|
418
|
+
displayName: z4.string(),
|
|
419
|
+
scopes: z4.array(tokenScopeSchema),
|
|
420
|
+
tokenLabel: z4.string().nullable()
|
|
421
|
+
});
|
|
422
|
+
var tokenCreateBodySchema = z4.object({
|
|
423
|
+
label: z4.string().min(1).max(120),
|
|
424
|
+
scopes: z4.array(tokenScopeSchema).min(1),
|
|
425
|
+
expiresAt: z4.string().datetime().optional()
|
|
426
|
+
});
|
|
427
|
+
var tokenSummarySchema = z4.object({
|
|
428
|
+
id: z4.string().uuid(),
|
|
429
|
+
tokenIdPrefix: z4.string(),
|
|
430
|
+
label: z4.string(),
|
|
431
|
+
displayName: z4.string(),
|
|
432
|
+
scopes: z4.array(tokenScopeSchema),
|
|
433
|
+
createdAt: z4.string().datetime(),
|
|
434
|
+
lastUsedAt: z4.string().datetime().nullable(),
|
|
435
|
+
expiresAt: z4.string().datetime().nullable()
|
|
436
|
+
});
|
|
437
|
+
var tokenCreatedSchema = tokenSummarySchema.extend({
|
|
438
|
+
tokenString: tokenStringSchema
|
|
439
|
+
});
|
|
440
|
+
var teamSummarySchema = z4.object({
|
|
441
|
+
id: z4.string().uuid(),
|
|
442
|
+
name: z4.string(),
|
|
443
|
+
createdAt: z4.string().datetime()
|
|
444
|
+
});
|
|
445
|
+
var teamCreateBodySchema = z4.object({
|
|
446
|
+
name: z4.string().min(1).max(120),
|
|
447
|
+
initialUserDisplayName: z4.string().min(1).max(120).default("Admin"),
|
|
448
|
+
initialToken: z4.object({
|
|
449
|
+
label: z4.string().min(1).max(120),
|
|
450
|
+
scopes: z4.array(tokenScopeSchema).min(1)
|
|
451
|
+
}).optional()
|
|
452
|
+
});
|
|
453
|
+
var teamCreatedSchema = teamSummarySchema.extend({
|
|
454
|
+
initialUser: z4.object({
|
|
455
|
+
id: z4.string().uuid(),
|
|
456
|
+
displayName: z4.string()
|
|
457
|
+
}),
|
|
458
|
+
initialToken: tokenCreatedSchema.nullable()
|
|
459
|
+
});
|
|
460
|
+
var setupStatusSchema = z4.object({
|
|
461
|
+
needsSetup: z4.boolean()
|
|
462
|
+
});
|
|
463
|
+
var setupBodySchema = z4.object({
|
|
464
|
+
teamName: z4.string().min(1).max(120),
|
|
465
|
+
displayName: z4.string().min(1).max(120).default("Admin")
|
|
466
|
+
});
|
|
467
|
+
var setupResultSchema = z4.object({
|
|
468
|
+
team: teamSummarySchema,
|
|
469
|
+
user: z4.object({ id: z4.string().uuid(), displayName: z4.string() }),
|
|
470
|
+
tokenString: tokenStringSchema,
|
|
471
|
+
scopes: z4.array(tokenScopeSchema)
|
|
472
|
+
});
|
|
473
|
+
var teamLlmSettingsSchema = z4.object({
|
|
474
|
+
keyConfigured: z4.boolean(),
|
|
475
|
+
keyHint: z4.string().nullable(),
|
|
476
|
+
lightModel: z4.string(),
|
|
477
|
+
heavyModel: z4.string()
|
|
478
|
+
});
|
|
479
|
+
var teamLlmUpdateSchema = z4.object({
|
|
480
|
+
openrouterApiKey: z4.string().max(400).nullable().optional(),
|
|
481
|
+
lightModel: z4.string().min(1).max(200).optional(),
|
|
482
|
+
heavyModel: z4.string().min(1).max(200).optional()
|
|
483
|
+
});
|
|
484
|
+
var wcvpImportStatusSchema = z4.enum([
|
|
485
|
+
"queued",
|
|
486
|
+
"downloading",
|
|
487
|
+
"importing",
|
|
488
|
+
"completed",
|
|
489
|
+
"failed",
|
|
490
|
+
"cancelled"
|
|
491
|
+
]);
|
|
492
|
+
var wcvpImportRunSchema = z4.object({
|
|
493
|
+
id: z4.string().uuid(),
|
|
494
|
+
version: z4.string(),
|
|
495
|
+
sourceUrl: z4.string(),
|
|
496
|
+
status: wcvpImportStatusSchema,
|
|
497
|
+
forceOverwrite: z4.boolean(),
|
|
498
|
+
bytesDownloaded: z4.number().int().nonnegative(),
|
|
499
|
+
totalBytes: z4.number().int().nonnegative().nullable(),
|
|
500
|
+
insertedCount: z4.number().int().nonnegative(),
|
|
501
|
+
recordCount: z4.number().int().nonnegative().nullable(),
|
|
502
|
+
message: z4.string().nullable(),
|
|
503
|
+
createdAt: z4.string().datetime(),
|
|
504
|
+
startedAt: z4.string().datetime().nullable(),
|
|
505
|
+
completedAt: z4.string().datetime().nullable()
|
|
506
|
+
});
|
|
507
|
+
var wcvpImportCreateBodySchema = z4.object({
|
|
508
|
+
version: z4.string().min(1).max(120),
|
|
509
|
+
/** Defaults to the Kew SFTP URL when omitted, so the admin doesn't have
|
|
510
|
+
* to remember it for routine v13/v14 imports. */
|
|
511
|
+
url: z4.string().url().optional(),
|
|
512
|
+
force: z4.boolean().optional()
|
|
513
|
+
});
|
|
514
|
+
var adminHealthErrorSchema = z4.object({
|
|
515
|
+
id: z4.string().uuid(),
|
|
516
|
+
action: z4.string(),
|
|
517
|
+
entityType: z4.string(),
|
|
518
|
+
entityId: z4.string().nullable(),
|
|
519
|
+
timestamp: z4.string().datetime(),
|
|
520
|
+
metadata: z4.unknown().nullable()
|
|
521
|
+
});
|
|
522
|
+
var adminHealthSchema = z4.object({
|
|
523
|
+
queueDepth: z4.object({
|
|
524
|
+
waiting: z4.number().int().nonnegative(),
|
|
525
|
+
active: z4.number().int().nonnegative(),
|
|
526
|
+
delayed: z4.number().int().nonnegative(),
|
|
527
|
+
completed: z4.number().int().nonnegative(),
|
|
528
|
+
failed: z4.number().int().nonnegative(),
|
|
529
|
+
paused: z4.number().int().nonnegative()
|
|
530
|
+
}),
|
|
531
|
+
workerCount: z4.number().int().nonnegative(),
|
|
532
|
+
lastErrors: z4.array(adminHealthErrorSchema),
|
|
533
|
+
rateLimitBudget: z4.object({
|
|
534
|
+
max: z4.number().int().nonnegative(),
|
|
535
|
+
timeWindowSeconds: z4.number().int().nonnegative()
|
|
536
|
+
}),
|
|
537
|
+
snapshotVersion: z4.string().nullable(),
|
|
538
|
+
snapshotImportedAt: z4.string().datetime().nullable(),
|
|
539
|
+
/**
|
|
540
|
+
* Number of cached accepted-name mappings (L0.5 match-history rows) for the
|
|
541
|
+
* caller's team — the size of the team accepted-name cache.
|
|
542
|
+
*/
|
|
543
|
+
teamHistorySize: z4.number().int().nonnegative(),
|
|
544
|
+
/**
|
|
545
|
+
* Overall (all-teams) tally of which cascade layer produced the winning
|
|
546
|
+
* candidate, across every matched query. `layer` is the raw attempt layer
|
|
547
|
+
* (`L1`…`L7`, `L0.5`, `OVERRIDE`); `count` is the number of matched queries
|
|
548
|
+
* that layer resolved. Descending by count.
|
|
549
|
+
*/
|
|
550
|
+
layerUsage: z4.array(
|
|
551
|
+
z4.object({
|
|
552
|
+
layer: z4.string(),
|
|
553
|
+
count: z4.number().int().nonnegative()
|
|
554
|
+
})
|
|
555
|
+
),
|
|
556
|
+
/**
|
|
557
|
+
* Per external provider (GBIF / TNRS / GNverifier / Pl@ntNet / OpenRouter)
|
|
558
|
+
* call stats across all teams, from `match_attempts`. Local/in-process
|
|
559
|
+
* providers (wcvp-local / gnparser / team-history / cascade) are excluded.
|
|
560
|
+
* Lets the health page surface each external source's usage, hit/miss/error
|
|
561
|
+
* breakdown, latency, and recency — so a misbehaving provider is visible.
|
|
562
|
+
*/
|
|
563
|
+
externalProviders: z4.array(
|
|
564
|
+
z4.object({
|
|
565
|
+
provider: z4.string(),
|
|
566
|
+
total: z4.number().int().nonnegative(),
|
|
567
|
+
hits: z4.number().int().nonnegative(),
|
|
568
|
+
misses: z4.number().int().nonnegative(),
|
|
569
|
+
errors: z4.number().int().nonnegative(),
|
|
570
|
+
avgDurationMs: z4.number().int().nonnegative(),
|
|
571
|
+
lastUsedAt: z4.string().datetime().nullable()
|
|
572
|
+
})
|
|
573
|
+
)
|
|
574
|
+
});
|
|
575
|
+
|
|
576
|
+
// ../shared/src/normalize.ts
|
|
577
|
+
var NULL_SENTINELS = /* @__PURE__ */ new Set(["", "na", "n/a", "null", "-", "\u2014", "unknown", "undet", "undet."]);
|
|
578
|
+
var QUALIFIER_PATTERNS = [
|
|
579
|
+
{ pattern: /\bcf\.?\b/i, qualifier: "cf" },
|
|
580
|
+
{ pattern: /\baff\.?\b/i, qualifier: "aff" },
|
|
581
|
+
{ pattern: /\bs\.\s*l\.?\b/i, qualifier: "sensu_lato" },
|
|
582
|
+
{ pattern: /\bs\.\s*str\.?\b/i, qualifier: "sensu_stricto" },
|
|
583
|
+
{ pattern: /\bs\.\s*s\.?\b/i, qualifier: "sensu_stricto" },
|
|
584
|
+
{ pattern: /\bagg\.?\b/i, qualifier: "aggregate" },
|
|
585
|
+
{ pattern: /\bcomplex\b/i, qualifier: "complex" },
|
|
586
|
+
{ pattern: /\bspp\.?\b/i, qualifier: "spp" },
|
|
587
|
+
{ pattern: /\bsp\.?(?=\s|$)/i, qualifier: "sp" },
|
|
588
|
+
{ pattern: /\bindet\.?\b/i, qualifier: "indet" }
|
|
589
|
+
];
|
|
590
|
+
var CULTIVAR_PATTERN = /'[^']+'|"[^"]+"/;
|
|
591
|
+
var HYBRID_FORMULA = /\w+\s+[×x]\s+[A-ZÀ-Ý]/;
|
|
592
|
+
function normalizeInput(raw) {
|
|
593
|
+
const original = (raw ?? "").toString();
|
|
594
|
+
const warnings = [];
|
|
595
|
+
let s = original.normalize("NFC").trim();
|
|
596
|
+
if (/\s{2,}/.test(s)) warnings.push("multiple_spaces");
|
|
597
|
+
s = s.replace(/\s+/g, " ");
|
|
598
|
+
if (s.startsWith('"') && s.endsWith('"') || s.startsWith("'") && s.endsWith("'")) {
|
|
599
|
+
s = s.slice(1, -1).trim();
|
|
600
|
+
}
|
|
601
|
+
if (NULL_SENTINELS.has(s.toLowerCase())) {
|
|
602
|
+
return {
|
|
603
|
+
raw: original,
|
|
604
|
+
normalized: null,
|
|
605
|
+
qualifier: "none",
|
|
606
|
+
isCultivar: false,
|
|
607
|
+
isHybridFormula: false,
|
|
608
|
+
isVernacularSuspected: false,
|
|
609
|
+
warnings
|
|
610
|
+
};
|
|
611
|
+
}
|
|
612
|
+
if (s.endsWith(".") || /[,;]$/.test(s)) warnings.push("trailing_punctuation");
|
|
613
|
+
if (s.includes("?")) warnings.push("question_mark");
|
|
614
|
+
s = s.replace(/(?<=\w)\s+[xX]\s+(?=\w)/g, " \xD7 ");
|
|
615
|
+
let qualifier = "none";
|
|
616
|
+
for (const { pattern, qualifier: q } of QUALIFIER_PATTERNS) {
|
|
617
|
+
if (pattern.test(s)) {
|
|
618
|
+
qualifier = q;
|
|
619
|
+
break;
|
|
620
|
+
}
|
|
621
|
+
}
|
|
622
|
+
const isCultivar = CULTIVAR_PATTERN.test(s);
|
|
623
|
+
if (isCultivar) qualifier = "cultivar";
|
|
624
|
+
const isHybridFormula = HYBRID_FORMULA.test(s);
|
|
625
|
+
if (isHybridFormula && qualifier === "none") qualifier = "hybrid_formula";
|
|
626
|
+
const isVernacularSuspected = detectVernacular(s);
|
|
627
|
+
return {
|
|
628
|
+
raw: original,
|
|
629
|
+
normalized: s,
|
|
630
|
+
qualifier,
|
|
631
|
+
isCultivar,
|
|
632
|
+
isHybridFormula,
|
|
633
|
+
isVernacularSuspected,
|
|
634
|
+
warnings
|
|
635
|
+
};
|
|
636
|
+
}
|
|
637
|
+
var VERNACULAR_KEYWORDS = /\b(tree|flower|grass|weed|herb|fern|moss|leaf|lily|rose|oak)\b/i;
|
|
638
|
+
function detectVernacular(s) {
|
|
639
|
+
if (!s) return false;
|
|
640
|
+
if (VERNACULAR_KEYWORDS.test(s)) return true;
|
|
641
|
+
const firstChar = s[0];
|
|
642
|
+
if (!firstChar) return false;
|
|
643
|
+
if (firstChar !== firstChar.toUpperCase()) return true;
|
|
644
|
+
return false;
|
|
645
|
+
}
|
|
646
|
+
|
|
647
|
+
// ../shared/src/id-type.ts
|
|
648
|
+
var IPNI_LSID_URN_RE = /^urn:lsid:ipni\.org:names:\d+-\d+$/i;
|
|
649
|
+
var IPNI_LSID_BARE_RE = /^\d+-\d+$/;
|
|
650
|
+
var WFO_RE = /^wfo-\d{10}$/i;
|
|
651
|
+
var INTEGER_RE = /^\d{4,}$/;
|
|
652
|
+
var COL_RE = /^[A-Z0-9]{5,6}$/;
|
|
653
|
+
var URN_RE = /^urn:/i;
|
|
654
|
+
var URL_RE = /^https?:\/\//i;
|
|
655
|
+
function detectIdType(value) {
|
|
656
|
+
const v = value.trim();
|
|
657
|
+
if (!v) return "unknown";
|
|
658
|
+
if (IPNI_LSID_URN_RE.test(v)) return "ipni-lsid-urn";
|
|
659
|
+
if (IPNI_LSID_BARE_RE.test(v)) return "ipni-lsid";
|
|
660
|
+
if (WFO_RE.test(v)) return "wfo-name-id";
|
|
661
|
+
if (URN_RE.test(v)) return "urn";
|
|
662
|
+
if (URL_RE.test(v)) return "url";
|
|
663
|
+
if (INTEGER_RE.test(v)) return "gbif-taxon-key";
|
|
664
|
+
if (COL_RE.test(v)) return "col-name-id";
|
|
665
|
+
if (v.length <= 3) return "short-string";
|
|
666
|
+
return "unknown";
|
|
667
|
+
}
|
|
668
|
+
function detectIdTypeDistribution(samples, opts) {
|
|
669
|
+
const counts = {
|
|
670
|
+
"ipni-lsid": 0,
|
|
671
|
+
"ipni-lsid-urn": 0,
|
|
672
|
+
"wcvp-taxon-id": 0,
|
|
673
|
+
"gbif-taxon-key": 0,
|
|
674
|
+
"wfo-name-id": 0,
|
|
675
|
+
"col-name-id": 0,
|
|
676
|
+
urn: 0,
|
|
677
|
+
url: 0,
|
|
678
|
+
"short-string": 0,
|
|
679
|
+
unknown: 0
|
|
680
|
+
};
|
|
681
|
+
const typed = [];
|
|
682
|
+
for (const s of samples) {
|
|
683
|
+
const t = detectIdType(s.value);
|
|
684
|
+
counts[t]++;
|
|
685
|
+
typed.push({ rowIndex: s.rowIndex, value: s.value, type: t });
|
|
686
|
+
}
|
|
687
|
+
let dominant = null;
|
|
688
|
+
let max = 0;
|
|
689
|
+
for (const [t, n] of Object.entries(counts)) {
|
|
690
|
+
if (n > max) {
|
|
691
|
+
max = n;
|
|
692
|
+
dominant = t;
|
|
693
|
+
}
|
|
694
|
+
}
|
|
695
|
+
const sampleSize = samples.length;
|
|
696
|
+
const dominantConfidence = sampleSize > 0 && dominant ? counts[dominant] / sampleSize : 0;
|
|
697
|
+
const minorityExamples = typed.filter((t) => t.type !== dominant).slice(0, opts?.minorityExamples ?? 5);
|
|
698
|
+
return { sampleSize, counts, dominant, dominantConfidence, minorityExamples };
|
|
699
|
+
}
|
|
700
|
+
|
|
701
|
+
// ../shared/src/column-map.ts
|
|
702
|
+
import { z as z5 } from "zod";
|
|
703
|
+
var columnMappingSchema = z5.object({
|
|
704
|
+
nameColumn: z5.string().nullable(),
|
|
705
|
+
idColumn: z5.string().nullable(),
|
|
706
|
+
familyColumn: z5.string().nullable(),
|
|
707
|
+
genusColumn: z5.string().nullable(),
|
|
708
|
+
rankColumn: z5.string().nullable(),
|
|
709
|
+
authorColumn: z5.string().nullable()
|
|
710
|
+
});
|
|
711
|
+
var detectColumnsBodySchema = z5.object({
|
|
712
|
+
headers: z5.array(z5.string().min(1)).min(1).max(200)
|
|
713
|
+
});
|
|
714
|
+
var detectColumnsResponseSchema = z5.object({
|
|
715
|
+
mapping: columnMappingSchema,
|
|
716
|
+
usedLlm: z5.boolean()
|
|
717
|
+
});
|
|
718
|
+
|
|
719
|
+
// src/api-client.ts
|
|
720
|
+
import { promises as fs } from "fs";
|
|
721
|
+
import { basename } from "path";
|
|
722
|
+
import { FormData, fetch, request } from "undici";
|
|
723
|
+
var ApiError = class extends Error {
|
|
724
|
+
constructor(message, status, body) {
|
|
725
|
+
super(message);
|
|
726
|
+
this.status = status;
|
|
727
|
+
this.body = body;
|
|
728
|
+
}
|
|
729
|
+
status;
|
|
730
|
+
body;
|
|
731
|
+
};
|
|
732
|
+
async function call(creds, path, init) {
|
|
733
|
+
const url = new URL(path, creds.server).toString();
|
|
734
|
+
const hasBody = init?.body !== void 0;
|
|
735
|
+
const res = await request(url, {
|
|
736
|
+
method: init?.method ?? "GET",
|
|
737
|
+
headers: {
|
|
738
|
+
authorization: `Bearer ${creds.token}`,
|
|
739
|
+
...hasBody ? { "content-type": "application/json" } : {}
|
|
740
|
+
},
|
|
741
|
+
...hasBody ? { body: JSON.stringify(init.body) } : {}
|
|
742
|
+
});
|
|
743
|
+
const text = await res.body.text();
|
|
744
|
+
const parsed = text ? safeJson(text) : null;
|
|
745
|
+
if (res.statusCode >= 400) {
|
|
746
|
+
throw new ApiError(`HTTP ${res.statusCode}`, res.statusCode, parsed ?? text);
|
|
747
|
+
}
|
|
748
|
+
return parsed;
|
|
749
|
+
}
|
|
750
|
+
function safeJson(s) {
|
|
751
|
+
try {
|
|
752
|
+
return JSON.parse(s);
|
|
753
|
+
} catch {
|
|
754
|
+
return null;
|
|
755
|
+
}
|
|
756
|
+
}
|
|
757
|
+
var apiClient = {
|
|
758
|
+
me: (c) => call(c, "/v1/me"),
|
|
759
|
+
listJobs: (c) => call(c, "/v1/jobs"),
|
|
760
|
+
getJob: (c, id) => call(c, `/v1/jobs/${id}`),
|
|
761
|
+
downloadColumns: (c, id) => call(c, `/v1/jobs/${id}/download/columns`),
|
|
762
|
+
pauseJob: (c, id) => call(c, `/v1/jobs/${id}/pause`, { method: "POST" }),
|
|
763
|
+
resumeJob: (c, id) => call(c, `/v1/jobs/${id}/resume`, { method: "POST" }),
|
|
764
|
+
cancelJob: (c, id) => call(c, `/v1/jobs/${id}/cancel`, { method: "POST" }),
|
|
765
|
+
async submitJob(creds, filePath, config, name) {
|
|
766
|
+
const buf = await fs.readFile(filePath);
|
|
767
|
+
const fd = new FormData();
|
|
768
|
+
const lower = filePath.toLowerCase();
|
|
769
|
+
const mime = lower.endsWith(".json") ? "application/json" : "text/csv";
|
|
770
|
+
fd.set("file", new Blob([buf], { type: mime }), basename(filePath));
|
|
771
|
+
fd.set("config", JSON.stringify(config));
|
|
772
|
+
if (name && name.trim()) fd.set("name", name.trim());
|
|
773
|
+
const url = new URL("/v1/jobs", creds.server).toString();
|
|
774
|
+
const res = await fetch(url, {
|
|
775
|
+
method: "POST",
|
|
776
|
+
body: fd,
|
|
777
|
+
headers: { authorization: `Bearer ${creds.token}` }
|
|
778
|
+
});
|
|
779
|
+
const text = await res.text();
|
|
780
|
+
const parsed = text ? safeJson(text) : null;
|
|
781
|
+
if (!res.ok) throw new ApiError(`HTTP ${res.status}`, res.status, parsed ?? text);
|
|
782
|
+
return parsed;
|
|
783
|
+
},
|
|
784
|
+
async downloadJob(creds, id, opts) {
|
|
785
|
+
const params = new URLSearchParams({ format: opts.format });
|
|
786
|
+
if (opts.confirmedOnly) params.set("confirmedOnly", "true");
|
|
787
|
+
if (opts.bundle) params.set("bundle", "true");
|
|
788
|
+
if (opts.columns) params.set("columns", opts.columns);
|
|
789
|
+
if (opts.wcvpExtra) params.set("wcvpExtra", opts.wcvpExtra);
|
|
790
|
+
const url = new URL(`/v1/jobs/${id}/download?${params}`, creds.server).toString();
|
|
791
|
+
const res = await fetch(url, { headers: { authorization: `Bearer ${creds.token}` } });
|
|
792
|
+
if (!res.ok) {
|
|
793
|
+
const text = await res.text().catch(() => "");
|
|
794
|
+
throw new ApiError(`HTTP ${res.status}`, res.status, text);
|
|
795
|
+
}
|
|
796
|
+
const disp = res.headers.get("content-disposition") ?? "";
|
|
797
|
+
const m = /filename="?([^"]+)"?/.exec(disp);
|
|
798
|
+
const ext = opts.bundle ? "zip" : opts.format === "csv" ? "csv" : opts.format === "json" ? "json" : "ndjson";
|
|
799
|
+
const filename = m?.[1] ?? `planttaxomatcher_${id}.${ext}`;
|
|
800
|
+
const buf = Buffer.from(await res.arrayBuffer());
|
|
801
|
+
return { filename, body: buf };
|
|
802
|
+
},
|
|
803
|
+
async *streamJob(creds, id) {
|
|
804
|
+
const url = new URL(`/v1/jobs/${id}/stream`, creds.server).toString();
|
|
805
|
+
const res = await fetch(url, { headers: { authorization: `Bearer ${creds.token}` } });
|
|
806
|
+
if (!res.ok || !res.body) {
|
|
807
|
+
const text = await res.text().catch(() => "");
|
|
808
|
+
throw new ApiError(`HTTP ${res.status}`, res.status, text);
|
|
809
|
+
}
|
|
810
|
+
const reader = res.body.getReader();
|
|
811
|
+
const decoder = new TextDecoder("utf-8");
|
|
812
|
+
let buf = "";
|
|
813
|
+
for (; ; ) {
|
|
814
|
+
const { done, value } = await reader.read();
|
|
815
|
+
if (done) break;
|
|
816
|
+
buf += decoder.decode(value, { stream: true });
|
|
817
|
+
let nl;
|
|
818
|
+
while ((nl = buf.indexOf("\n")) >= 0) {
|
|
819
|
+
const line = buf.slice(0, nl).trim();
|
|
820
|
+
buf = buf.slice(nl + 1);
|
|
821
|
+
if (!line) continue;
|
|
822
|
+
try {
|
|
823
|
+
yield JSON.parse(line);
|
|
824
|
+
} catch {
|
|
825
|
+
}
|
|
826
|
+
}
|
|
827
|
+
}
|
|
828
|
+
}
|
|
829
|
+
};
|
|
830
|
+
|
|
831
|
+
// src/config.ts
|
|
832
|
+
import { promises as fs2 } from "fs";
|
|
833
|
+
import { homedir } from "os";
|
|
834
|
+
import { join } from "path";
|
|
835
|
+
var CONFIG_DIR = join(homedir(), ".config", "planttaxomatcher");
|
|
836
|
+
var CONFIG_FILE = join(CONFIG_DIR, "credentials");
|
|
837
|
+
async function readCredentials() {
|
|
838
|
+
try {
|
|
839
|
+
const raw = await fs2.readFile(CONFIG_FILE, "utf8");
|
|
840
|
+
return JSON.parse(raw);
|
|
841
|
+
} catch (err) {
|
|
842
|
+
if (err.code === "ENOENT") return null;
|
|
843
|
+
throw err;
|
|
844
|
+
}
|
|
845
|
+
}
|
|
846
|
+
async function writeCredentials(creds) {
|
|
847
|
+
await fs2.mkdir(CONFIG_DIR, { recursive: true, mode: 448 });
|
|
848
|
+
await fs2.writeFile(CONFIG_FILE, JSON.stringify(creds, null, 2), { mode: 384 });
|
|
849
|
+
}
|
|
850
|
+
async function clearCredentials() {
|
|
851
|
+
try {
|
|
852
|
+
await fs2.unlink(CONFIG_FILE);
|
|
853
|
+
} catch (err) {
|
|
854
|
+
if (err.code !== "ENOENT") throw err;
|
|
855
|
+
}
|
|
856
|
+
}
|
|
857
|
+
async function requireCredentials() {
|
|
858
|
+
const creds = await readCredentials();
|
|
859
|
+
if (!creds) {
|
|
860
|
+
throw new Error("Not logged in. Run: planttaxomatcher login --token <t> --server <url>");
|
|
861
|
+
}
|
|
862
|
+
return creds;
|
|
863
|
+
}
|
|
864
|
+
|
|
865
|
+
// src/dry-run.ts
|
|
866
|
+
import { promises as fs3 } from "fs";
|
|
867
|
+
import { Readable } from "stream";
|
|
868
|
+
import { parse as parseCsvStream } from "csv-parse";
|
|
869
|
+
import kleur from "kleur";
|
|
870
|
+
async function readSample(file, sampleLimit) {
|
|
871
|
+
const lower = file.toLowerCase();
|
|
872
|
+
if (lower.endsWith(".json")) {
|
|
873
|
+
const text = await fs3.readFile(file, "utf8");
|
|
874
|
+
const arr = JSON.parse(text);
|
|
875
|
+
if (!Array.isArray(arr)) throw new Error("JSON file must be an array of row objects");
|
|
876
|
+
const out2 = [];
|
|
877
|
+
for (const v of arr) {
|
|
878
|
+
if (out2.length >= sampleLimit) break;
|
|
879
|
+
if (v && typeof v === "object") {
|
|
880
|
+
const row = {};
|
|
881
|
+
for (const [k, val] of Object.entries(v)) row[k] = val == null ? "" : String(val);
|
|
882
|
+
out2.push(row);
|
|
883
|
+
}
|
|
884
|
+
}
|
|
885
|
+
return out2;
|
|
886
|
+
}
|
|
887
|
+
const head = Buffer.alloc(4096);
|
|
888
|
+
const fh = await fs3.open(file, "r");
|
|
889
|
+
let bytesRead = 0;
|
|
890
|
+
try {
|
|
891
|
+
bytesRead = (await fh.read(head, 0, 4096, 0)).bytesRead;
|
|
892
|
+
} finally {
|
|
893
|
+
await fh.close();
|
|
894
|
+
}
|
|
895
|
+
const delimiter = detectDelimiter(head.subarray(0, bytesRead));
|
|
896
|
+
const buf = await fs3.readFile(file);
|
|
897
|
+
const parser = Readable.from(buf).pipe(
|
|
898
|
+
parseCsvStream({
|
|
899
|
+
columns: true,
|
|
900
|
+
skip_empty_lines: true,
|
|
901
|
+
trim: true,
|
|
902
|
+
delimiter,
|
|
903
|
+
relax_quotes: true,
|
|
904
|
+
relax_column_count: true
|
|
905
|
+
})
|
|
906
|
+
);
|
|
907
|
+
const out = [];
|
|
908
|
+
for await (const row of parser) {
|
|
909
|
+
out.push(row);
|
|
910
|
+
if (out.length >= sampleLimit) break;
|
|
911
|
+
}
|
|
912
|
+
return out;
|
|
913
|
+
}
|
|
914
|
+
function detectDelimiter(buf) {
|
|
915
|
+
const head = buf.toString("utf8").split("\n", 5).join("\n");
|
|
916
|
+
const counts = {
|
|
917
|
+
",": (head.match(/,/g) ?? []).length,
|
|
918
|
+
";": (head.match(/;/g) ?? []).length,
|
|
919
|
+
" ": (head.match(/\t/g) ?? []).length,
|
|
920
|
+
"|": (head.match(/\|/g) ?? []).length
|
|
921
|
+
};
|
|
922
|
+
const winner = Object.entries(counts).sort((a, b) => b[1] - a[1])[0];
|
|
923
|
+
return winner && winner[1] > 0 ? winner[0] : ",";
|
|
924
|
+
}
|
|
925
|
+
async function buildDryRunReport(file, opts) {
|
|
926
|
+
const sampleLimit = opts.sampleLimit ?? 1e3;
|
|
927
|
+
const previewRows = opts.previewRows ?? 10;
|
|
928
|
+
const rows = await readSample(file, sampleLimit);
|
|
929
|
+
if (rows.length === 0) {
|
|
930
|
+
return {
|
|
931
|
+
rowsRead: 0,
|
|
932
|
+
nullNameCount: 0,
|
|
933
|
+
qualifierFlagCount: 0,
|
|
934
|
+
sampleRows: [],
|
|
935
|
+
idTypeDetection: null
|
|
936
|
+
};
|
|
937
|
+
}
|
|
938
|
+
if (!(opts.nameColumn in rows[0])) {
|
|
939
|
+
throw new Error(`name column "${opts.nameColumn}" not present in file header`);
|
|
940
|
+
}
|
|
941
|
+
let nullNameCount = 0;
|
|
942
|
+
let qualifierFlagCount = 0;
|
|
943
|
+
const sampleRows = [];
|
|
944
|
+
for (let i = 0; i < rows.length; i++) {
|
|
945
|
+
const row = rows[i];
|
|
946
|
+
const input = row[opts.nameColumn] ?? null;
|
|
947
|
+
const parsed = normalizeInput(input);
|
|
948
|
+
if (!parsed.normalized) nullNameCount++;
|
|
949
|
+
const flags = [];
|
|
950
|
+
if (parsed.qualifier && parsed.qualifier !== "none") flags.push(`qualifier:${parsed.qualifier}`);
|
|
951
|
+
if (parsed.isCultivar) flags.push("cultivar");
|
|
952
|
+
if (parsed.isHybridFormula) flags.push("hybrid-formula");
|
|
953
|
+
if (parsed.isVernacularSuspected) flags.push("vernacular?");
|
|
954
|
+
if (flags.length > 0) qualifierFlagCount++;
|
|
955
|
+
if (sampleRows.length < previewRows) {
|
|
956
|
+
sampleRows.push({
|
|
957
|
+
rowIndex: i,
|
|
958
|
+
input,
|
|
959
|
+
normalized: parsed.normalized,
|
|
960
|
+
flags
|
|
961
|
+
});
|
|
962
|
+
}
|
|
963
|
+
}
|
|
964
|
+
let idTypeDetection = null;
|
|
965
|
+
if (opts.idColumn && rows.length > 0 && opts.idColumn in rows[0]) {
|
|
966
|
+
const samples = rows.map((r, i) => ({ rowIndex: i, value: r[opts.idColumn] ?? "" })).filter((s) => s.value.trim().length > 0);
|
|
967
|
+
if (samples.length > 0) {
|
|
968
|
+
idTypeDetection = detectIdTypeDistribution(samples, { minorityExamples: 5 });
|
|
969
|
+
}
|
|
970
|
+
}
|
|
971
|
+
return {
|
|
972
|
+
rowsRead: rows.length,
|
|
973
|
+
nullNameCount,
|
|
974
|
+
qualifierFlagCount,
|
|
975
|
+
sampleRows,
|
|
976
|
+
idTypeDetection
|
|
977
|
+
};
|
|
978
|
+
}
|
|
979
|
+
function printDryRunReport(report) {
|
|
980
|
+
console.log();
|
|
981
|
+
console.log(kleur.bold("Dry-run preview"));
|
|
982
|
+
console.log(
|
|
983
|
+
` Read ${kleur.cyan(report.rowsRead)} rows \xB7 ${kleur.yellow(report.nullNameCount)} with no usable name \xB7 ${kleur.yellow(report.qualifierFlagCount)} with qualifier flags`
|
|
984
|
+
);
|
|
985
|
+
if (report.idTypeDetection) {
|
|
986
|
+
const d = report.idTypeDetection;
|
|
987
|
+
const pct = Math.round(d.dominantConfidence * 100);
|
|
988
|
+
console.log(
|
|
989
|
+
` ID-column detection: ${kleur.green(d.dominant ?? "\u2014")} (${pct}% of ${d.sampleSize} sampled)`
|
|
990
|
+
);
|
|
991
|
+
if (d.minorityExamples.length > 0) {
|
|
992
|
+
console.log(` Minority examples:`);
|
|
993
|
+
for (const m of d.minorityExamples) {
|
|
994
|
+
console.log(
|
|
995
|
+
` row ${m.rowIndex + 1}: ${kleur.gray(m.value)} \u2192 ${kleur.dim(m.type)}`
|
|
996
|
+
);
|
|
997
|
+
}
|
|
998
|
+
}
|
|
999
|
+
}
|
|
1000
|
+
console.log();
|
|
1001
|
+
console.log(kleur.bold("First rows after normalization:"));
|
|
1002
|
+
console.log(
|
|
1003
|
+
` ${"#".padStart(4)} ${"input".padEnd(36)} ${"normalized".padEnd(36)} flags`
|
|
1004
|
+
);
|
|
1005
|
+
for (const r of report.sampleRows) {
|
|
1006
|
+
const idx = String(r.rowIndex + 1).padStart(4);
|
|
1007
|
+
const inp = trunc(r.input ?? "\u2205", 36).padEnd(36);
|
|
1008
|
+
const norm = trunc(r.normalized ?? "\u2205", 36).padEnd(36);
|
|
1009
|
+
console.log(` ${idx} ${inp} ${norm} ${r.flags.join(" ") || ""}`);
|
|
1010
|
+
}
|
|
1011
|
+
console.log();
|
|
1012
|
+
}
|
|
1013
|
+
function trunc(s, n) {
|
|
1014
|
+
if (s.length <= n) return s;
|
|
1015
|
+
return s.slice(0, n - 1) + "\u2026";
|
|
1016
|
+
}
|
|
1017
|
+
|
|
1018
|
+
// src/index.ts
|
|
1019
|
+
var program = new Command();
|
|
1020
|
+
program.name("planttaxomatcher").description("PlantTaxoMatcher CLI").version("0.1.0");
|
|
1021
|
+
program.command("login").description("Save a personal token + server URL").option(
|
|
1022
|
+
"--token <token>",
|
|
1023
|
+
"Personal token (ptm_...). Avoid on shared hosts: it is visible in shell history and the process list. Prefer --token-stdin or the PLANTTAXOMATCHER_TOKEN env var."
|
|
1024
|
+
).option(
|
|
1025
|
+
"--token-stdin",
|
|
1026
|
+
"Read the token from stdin instead of argv (e.g. `cat token.txt | planttaxomatcher login --token-stdin`).",
|
|
1027
|
+
false
|
|
1028
|
+
).option("--server <url>", "API server base URL", "http://localhost:4000").option(
|
|
1029
|
+
"--insecure",
|
|
1030
|
+
"Allow sending the token over cleartext http to a non-loopback server (NOT recommended).",
|
|
1031
|
+
false
|
|
1032
|
+
).action(
|
|
1033
|
+
async (opts) => {
|
|
1034
|
+
assertServerTransport(opts.server, !!opts.insecure);
|
|
1035
|
+
const token = await resolveLoginToken(opts);
|
|
1036
|
+
if (!tokenStringSchema.safeParse(token).success) {
|
|
1037
|
+
throw new Error("malformed token \u2014 expected ptm_<12 chars>_<32 chars>");
|
|
1038
|
+
}
|
|
1039
|
+
await writeCredentials({ token, server: opts.server });
|
|
1040
|
+
console.log(kleur2.green("\u2713"), `Logged in to ${opts.server}`);
|
|
1041
|
+
}
|
|
1042
|
+
);
|
|
1043
|
+
program.command("logout").description("Clear saved credentials").action(async () => {
|
|
1044
|
+
await clearCredentials();
|
|
1045
|
+
console.log(kleur2.green("\u2713"), "Logged out");
|
|
1046
|
+
});
|
|
1047
|
+
program.command("whoami").description("Show current user").action(async () => {
|
|
1048
|
+
const creds = await requireCredentials();
|
|
1049
|
+
const me = await apiClient.me(creds);
|
|
1050
|
+
console.log(`${me.displayName} team=${me.teamId} scopes=${me.scopes.join(",")}`);
|
|
1051
|
+
});
|
|
1052
|
+
program.command("list").description("List recent jobs").action(async () => {
|
|
1053
|
+
const creds = await requireCredentials();
|
|
1054
|
+
const jobs = await apiClient.listJobs(creds);
|
|
1055
|
+
if (jobs.length === 0) {
|
|
1056
|
+
console.log(kleur2.gray("No jobs."));
|
|
1057
|
+
return;
|
|
1058
|
+
}
|
|
1059
|
+
for (const j of jobs) {
|
|
1060
|
+
console.log(
|
|
1061
|
+
`${j.id} ${j.status.padEnd(10)} ${String(j.matchedRows).padStart(6)}/${String(j.totalRows).padStart(6)} matched`
|
|
1062
|
+
);
|
|
1063
|
+
}
|
|
1064
|
+
});
|
|
1065
|
+
program.command("status <jobId>").description("Show job status snapshot").action(async (jobId) => {
|
|
1066
|
+
const creds = await requireCredentials();
|
|
1067
|
+
const job = await apiClient.getJob(creds, jobId);
|
|
1068
|
+
console.log(JSON.stringify(job, null, 2));
|
|
1069
|
+
});
|
|
1070
|
+
program.command("submit <files...>").description(
|
|
1071
|
+
'Submit one or more CSV/JSON files for matching. Accepts shell-expanded paths or quoted glob patterns (e.g. "data/*.csv"). Each file becomes its own job; the file name (without extension) is used as the job name.'
|
|
1072
|
+
).option(
|
|
1073
|
+
"--name <label>",
|
|
1074
|
+
"Job name override (single file only; ignored when multiple files match \u2014 the file name is used)"
|
|
1075
|
+
).requiredOption("--name-column <name>", "Column holding the scientific name").option("--id-column <name>", "Column holding a known ID").option("--family-column <name>", "Column holding family").option("--genus-column <name>", "Column holding genus").option("--rank-column <name>", "Column holding rank").option("--author-column <name>", "Column holding authorship").option("--author-mode <mode>", "ignore|prefer|strict", "prefer").option("--parallel <n>", "Per-job parallelism", "4").option("--review-mode <mode>", "off|recommended|strict", "recommended").option(
|
|
1076
|
+
"--keep-infraspecific",
|
|
1077
|
+
"Keep infraspecific accepted taxa (varieties, subspecies, forms) instead of collapsing them up to the species",
|
|
1078
|
+
false
|
|
1079
|
+
).option("--allow-llm", "Allow Layer 7 LLM cascade (breaks ties between competing matches)", false).option("--llm-cap-cents <cents>", "Max LLM spend in cents", "500").option("--referential <version>", "WCVP snapshot version to match against (default: latest)").option("--no-watch", "Do not stream progress after submit").option(
|
|
1080
|
+
"--dry-run",
|
|
1081
|
+
"Preview locally (normalize first rows + detect ID type) and confirm before uploading",
|
|
1082
|
+
false
|
|
1083
|
+
).option(
|
|
1084
|
+
"--dry-run-rows <n>",
|
|
1085
|
+
"Number of rows to show in the dry-run preview table (default 10)",
|
|
1086
|
+
"10"
|
|
1087
|
+
).action(async (files, opts) => {
|
|
1088
|
+
const creds = await requireCredentials();
|
|
1089
|
+
const inputs = await expandInputs(files);
|
|
1090
|
+
if (inputs.length === 0) throw new Error("no input files");
|
|
1091
|
+
if (opts.name && inputs.length > 1) {
|
|
1092
|
+
console.log(
|
|
1093
|
+
kleur2.yellow("!"),
|
|
1094
|
+
"--name ignored for multi-file submit; using each file name as the job name"
|
|
1095
|
+
);
|
|
1096
|
+
}
|
|
1097
|
+
if (inputs.length > 1) {
|
|
1098
|
+
console.log(kleur2.cyan("\u2192"), `${inputs.length} files matched:`);
|
|
1099
|
+
for (const f of inputs) console.log(kleur2.gray(` ${f}`));
|
|
1100
|
+
}
|
|
1101
|
+
if (opts.dryRun) {
|
|
1102
|
+
const idColumn = opts.idColumn ? String(opts.idColumn) : null;
|
|
1103
|
+
const previewRows = Number(opts.dryRunRows ?? 10);
|
|
1104
|
+
for (const file of inputs) {
|
|
1105
|
+
if (inputs.length > 1) console.log(kleur2.bold(`
|
|
1106
|
+
${file}`));
|
|
1107
|
+
const report = await buildDryRunReport(file, {
|
|
1108
|
+
nameColumn: String(opts.nameColumn),
|
|
1109
|
+
idColumn,
|
|
1110
|
+
previewRows: Number.isFinite(previewRows) ? previewRows : 10,
|
|
1111
|
+
sampleLimit: 1e3
|
|
1112
|
+
});
|
|
1113
|
+
printDryRunReport(report);
|
|
1114
|
+
}
|
|
1115
|
+
const proceed = await confirm({
|
|
1116
|
+
message: inputs.length > 1 ? `Proceed with upload of ${inputs.length} files?` : "Proceed with upload?",
|
|
1117
|
+
default: true
|
|
1118
|
+
});
|
|
1119
|
+
if (!proceed) {
|
|
1120
|
+
console.log(kleur2.gray("Aborted."));
|
|
1121
|
+
return;
|
|
1122
|
+
}
|
|
1123
|
+
}
|
|
1124
|
+
const config = {
|
|
1125
|
+
nameColumn: String(opts.nameColumn),
|
|
1126
|
+
idColumn: opts.idColumn ? String(opts.idColumn) : null,
|
|
1127
|
+
familyColumn: opts.familyColumn ? String(opts.familyColumn) : null,
|
|
1128
|
+
genusColumn: opts.genusColumn ? String(opts.genusColumn) : null,
|
|
1129
|
+
rankColumn: opts.rankColumn ? String(opts.rankColumn) : null,
|
|
1130
|
+
authorColumn: opts.authorColumn ? String(opts.authorColumn) : null,
|
|
1131
|
+
authorMode: String(opts.authorMode ?? "prefer"),
|
|
1132
|
+
matchAuthors: true,
|
|
1133
|
+
parallelism: Number(opts.parallel ?? 4),
|
|
1134
|
+
allowFuzzy: true,
|
|
1135
|
+
allowLlm: !!opts.allowLlm,
|
|
1136
|
+
llmCostCapCents: Number(opts.llmCapCents ?? 500),
|
|
1137
|
+
reviewMode: String(opts.reviewMode ?? "recommended"),
|
|
1138
|
+
// UI/CLI opt-in inverts the config flag: by default we collapse an
|
|
1139
|
+
// infraspecific accepted taxon up to its species.
|
|
1140
|
+
speciesLevelAcceptedOnly: !opts.keepInfraspecific,
|
|
1141
|
+
exportConfirmedOnly: false,
|
|
1142
|
+
forceReviewFamilies: [],
|
|
1143
|
+
...opts.referential ? { referentialVersion: String(opts.referential) } : {}
|
|
1144
|
+
};
|
|
1145
|
+
const submitted = [];
|
|
1146
|
+
for (const file of inputs) {
|
|
1147
|
+
const jobName = inputs.length === 1 && opts.name ? String(opts.name) : parsePath(file).name;
|
|
1148
|
+
console.log(kleur2.cyan("\u2192"), `Uploading ${file} \u2026`);
|
|
1149
|
+
const job = await apiClient.submitJob(creds, file, config, jobName || null);
|
|
1150
|
+
console.log(kleur2.green("\u2713"), `Job created: ${job.id} ${kleur2.gray(jobName)}`);
|
|
1151
|
+
console.log(` rows=${job.totalRows} uniqueQueries=${job.uniqueQueries}`);
|
|
1152
|
+
submitted.push({ id: job.id, file });
|
|
1153
|
+
}
|
|
1154
|
+
if (opts.watch === false) return;
|
|
1155
|
+
for (const s of submitted) {
|
|
1156
|
+
if (submitted.length > 1) console.log(kleur2.bold(`
|
|
1157
|
+
[${s.file}] ${s.id}`));
|
|
1158
|
+
await streamJob(creds, s.id);
|
|
1159
|
+
}
|
|
1160
|
+
});
|
|
1161
|
+
program.command("watch <jobId>").description("Stream NDJSON progress for a job").action(async (jobId) => {
|
|
1162
|
+
const creds = await requireCredentials();
|
|
1163
|
+
await streamJob(creds, jobId);
|
|
1164
|
+
});
|
|
1165
|
+
program.command("pause <jobId>").description("Pause a running job (worker stops between match queries)").action(async (jobId) => {
|
|
1166
|
+
const creds = await requireCredentials();
|
|
1167
|
+
const job = await apiClient.pauseJob(creds, jobId);
|
|
1168
|
+
console.log(kleur2.green("\u2713"), `paused: ${job.id} (status=${job.status})`);
|
|
1169
|
+
});
|
|
1170
|
+
program.command("resume <jobId>").description("Resume a paused job (re-enqueues the match stage)").action(async (jobId) => {
|
|
1171
|
+
const creds = await requireCredentials();
|
|
1172
|
+
const job = await apiClient.resumeJob(creds, jobId);
|
|
1173
|
+
console.log(kleur2.green("\u2713"), `resumed: ${job.id} (status=${job.status})`);
|
|
1174
|
+
});
|
|
1175
|
+
program.command("cancel <jobId>").description("Cancel a job. Already-matched rows are kept; pending queries stop.").action(async (jobId) => {
|
|
1176
|
+
const creds = await requireCredentials();
|
|
1177
|
+
const job = await apiClient.cancelJob(creds, jobId);
|
|
1178
|
+
console.log(kleur2.green("\u2713"), `cancelled: ${job.id} (status=${job.status})`);
|
|
1179
|
+
});
|
|
1180
|
+
program.command("download <jobId>").description("Download a job export (CSV, JSON, or NDJSON; optionally bundled with NOTICE.md)").option("--format <fmt>", "csv | json | ndjson", "csv").option("--confirmed-only", "only matched rows that are accepted / not pending", false).option("--bundle", "wrap in a ZIP with NOTICE.md citing the WCVP snapshot + providers", false).option(
|
|
1181
|
+
"--columns <list>",
|
|
1182
|
+
"comma-separated result/upload column keys to KEEP (default: all). See --list-columns"
|
|
1183
|
+
).option(
|
|
1184
|
+
"--wcvp-extra <list>",
|
|
1185
|
+
"comma-separated extra WCVP fields to append as wcvp_<key> columns (e.g. ipni_id,powo_id). See --list-columns"
|
|
1186
|
+
).option("--list-columns", "print the columns available for this job and exit", false).option("--output <path>", "write to this path; default is the server-provided filename in CWD").action(async (jobId, opts) => {
|
|
1187
|
+
const creds = await requireCredentials();
|
|
1188
|
+
if (opts.listColumns) {
|
|
1189
|
+
const cat = await apiClient.downloadColumns(creds, jobId);
|
|
1190
|
+
console.log(kleur2.bold("Result columns (--columns):"));
|
|
1191
|
+
for (const r of cat.result) console.log(` ${r.key} ${kleur2.gray(`(${r.group})`)}`);
|
|
1192
|
+
if (cat.original.length > 0) {
|
|
1193
|
+
console.log(kleur2.bold("\nYour upload columns (--columns):"));
|
|
1194
|
+
for (const k of cat.original) console.log(` ${k}`);
|
|
1195
|
+
}
|
|
1196
|
+
console.log(kleur2.bold("\nWCVP extra fields (--wcvp-extra):"));
|
|
1197
|
+
for (const f of cat.wcvpExtra) console.log(` ${f.key} ${kleur2.gray(`(${f.group})`)}`);
|
|
1198
|
+
return;
|
|
1199
|
+
}
|
|
1200
|
+
const format = opts.format ?? "csv";
|
|
1201
|
+
if (format !== "csv" && format !== "json" && format !== "ndjson") {
|
|
1202
|
+
throw new Error(`--format must be csv, json or ndjson (got ${String(format)})`);
|
|
1203
|
+
}
|
|
1204
|
+
const confirmedOnly = !!opts.confirmedOnly;
|
|
1205
|
+
const bundle = !!opts.bundle;
|
|
1206
|
+
const { filename, body } = await apiClient.downloadJob(creds, jobId, {
|
|
1207
|
+
format,
|
|
1208
|
+
confirmedOnly,
|
|
1209
|
+
bundle,
|
|
1210
|
+
...opts.columns ? { columns: String(opts.columns) } : {},
|
|
1211
|
+
...opts.wcvpExtra ? { wcvpExtra: String(opts.wcvpExtra) } : {}
|
|
1212
|
+
});
|
|
1213
|
+
const outPath = opts.output ?? filename;
|
|
1214
|
+
const { writeFile } = await import("fs/promises");
|
|
1215
|
+
await writeFile(outPath, body);
|
|
1216
|
+
console.log(kleur2.green("\u2713"), `wrote ${body.length} bytes to ${outPath}`);
|
|
1217
|
+
});
|
|
1218
|
+
program.parseAsync(process.argv).catch((err) => {
|
|
1219
|
+
const msg = err instanceof Error ? err.message : String(err);
|
|
1220
|
+
console.error(kleur2.red("error:"), msg);
|
|
1221
|
+
process.exit(1);
|
|
1222
|
+
});
|
|
1223
|
+
function assertServerTransport(server, insecure) {
|
|
1224
|
+
let u;
|
|
1225
|
+
try {
|
|
1226
|
+
u = new URL(server);
|
|
1227
|
+
} catch {
|
|
1228
|
+
throw new Error(`invalid --server URL: ${server}`);
|
|
1229
|
+
}
|
|
1230
|
+
if (u.protocol === "https:") return;
|
|
1231
|
+
if (u.protocol !== "http:") {
|
|
1232
|
+
throw new Error(`--server must use http or https (got ${u.protocol})`);
|
|
1233
|
+
}
|
|
1234
|
+
const host = u.hostname;
|
|
1235
|
+
const isLoopback = host === "localhost" || host === "127.0.0.1" || host === "::1" || host === "[::1]" || host.endsWith(".localhost");
|
|
1236
|
+
if (!isLoopback && !insecure) {
|
|
1237
|
+
throw new Error(
|
|
1238
|
+
`refusing to send a token in cleartext to ${u.host}. Use an https URL, or pass --insecure to override (NOT recommended).`
|
|
1239
|
+
);
|
|
1240
|
+
}
|
|
1241
|
+
}
|
|
1242
|
+
async function readStdinLine() {
|
|
1243
|
+
const chunks = [];
|
|
1244
|
+
for await (const chunk of process.stdin) chunks.push(chunk);
|
|
1245
|
+
return Buffer.concat(chunks).toString("utf8").trim();
|
|
1246
|
+
}
|
|
1247
|
+
async function resolveLoginToken(opts) {
|
|
1248
|
+
if (opts.tokenStdin) {
|
|
1249
|
+
const t = await readStdinLine();
|
|
1250
|
+
if (!t) throw new Error("--token-stdin was set but stdin was empty");
|
|
1251
|
+
return t;
|
|
1252
|
+
}
|
|
1253
|
+
if (opts.token) return opts.token.trim();
|
|
1254
|
+
const env = process.env.PLANTTAXOMATCHER_TOKEN;
|
|
1255
|
+
if (env && env.trim()) return env.trim();
|
|
1256
|
+
if (!process.stdin.isTTY) {
|
|
1257
|
+
throw new Error(
|
|
1258
|
+
"no token provided. Pass --token-stdin, set PLANTTAXOMATCHER_TOKEN, or run interactively."
|
|
1259
|
+
);
|
|
1260
|
+
}
|
|
1261
|
+
const entered = await password({ message: "Personal token (ptm_...)", mask: true });
|
|
1262
|
+
return entered.trim();
|
|
1263
|
+
}
|
|
1264
|
+
var GLOB_MAGIC = /[*?[\]{}!()]/;
|
|
1265
|
+
async function expandInputs(patterns) {
|
|
1266
|
+
const out = /* @__PURE__ */ new Set();
|
|
1267
|
+
for (const p of patterns) {
|
|
1268
|
+
if (GLOB_MAGIC.test(p)) {
|
|
1269
|
+
let matched = false;
|
|
1270
|
+
for await (const m of fs4.glob(p)) {
|
|
1271
|
+
out.add(m);
|
|
1272
|
+
matched = true;
|
|
1273
|
+
}
|
|
1274
|
+
if (!matched) throw new Error(`no files matched: ${p}`);
|
|
1275
|
+
} else {
|
|
1276
|
+
await fs4.access(p).catch(() => {
|
|
1277
|
+
throw new Error(`file not found: ${p}`);
|
|
1278
|
+
});
|
|
1279
|
+
out.add(p);
|
|
1280
|
+
}
|
|
1281
|
+
}
|
|
1282
|
+
return [...out].sort();
|
|
1283
|
+
}
|
|
1284
|
+
async function streamJob(creds, jobId) {
|
|
1285
|
+
console.log(kleur2.cyan("\u2192"), `Streaming progress for ${jobId} \u2026`);
|
|
1286
|
+
for await (const evt of apiClient.streamJob(creds, jobId)) {
|
|
1287
|
+
const t = String(evt.type ?? "");
|
|
1288
|
+
if (t === "heartbeat") continue;
|
|
1289
|
+
if (t === "status") {
|
|
1290
|
+
console.log(kleur2.gray("\u2022"), evt.status);
|
|
1291
|
+
} else if (t === "progress") {
|
|
1292
|
+
const p = evt.processedQueries ?? evt.processedRows ?? 0;
|
|
1293
|
+
const total = evt.totalQueries ?? evt.totalRows ?? 0;
|
|
1294
|
+
process.stdout.write(`\r progress: ${p}/${total} `);
|
|
1295
|
+
} else if (t === "completed") {
|
|
1296
|
+
process.stdout.write("\n");
|
|
1297
|
+
console.log(kleur2.green("\u2713"), "completed");
|
|
1298
|
+
return;
|
|
1299
|
+
} else if (t === "error") {
|
|
1300
|
+
process.stdout.write("\n");
|
|
1301
|
+
console.error(kleur2.red("error:"), evt.message);
|
|
1302
|
+
return;
|
|
1303
|
+
}
|
|
1304
|
+
}
|
|
1305
|
+
}
|
package/package.json
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "@plantnet/planttaxomatcher",
|
|
3
|
+
"version": "0.1.0",
|
|
4
|
+
"description": "PlantTaxoMatcher CLI — reconcile plant names against WCVP via the PlantTaxoMatcher API.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"engines": {
|
|
7
|
+
"node": ">=24"
|
|
8
|
+
},
|
|
9
|
+
"bin": {
|
|
10
|
+
"planttaxomatcher": "./dist/index.js"
|
|
11
|
+
},
|
|
12
|
+
"files": [
|
|
13
|
+
"dist"
|
|
14
|
+
],
|
|
15
|
+
"publishConfig": {
|
|
16
|
+
"access": "public"
|
|
17
|
+
},
|
|
18
|
+
"dependencies": {
|
|
19
|
+
"@inquirer/prompts": "^8.5.0",
|
|
20
|
+
"commander": "^14.0.0",
|
|
21
|
+
"csv-parse": "^6.0.0",
|
|
22
|
+
"kleur": "^4.1.5",
|
|
23
|
+
"ora": "^9.0.0",
|
|
24
|
+
"undici": "^8.0.0",
|
|
25
|
+
"zod": "^4.0.0"
|
|
26
|
+
},
|
|
27
|
+
"devDependencies": {
|
|
28
|
+
"@types/node": "^24.12.4",
|
|
29
|
+
"tsup": "^8.5.1",
|
|
30
|
+
"tsx": "^4.20.0",
|
|
31
|
+
"typescript": "^6.0.0",
|
|
32
|
+
"vitest": "^4.1.7",
|
|
33
|
+
"@planttaxomatcher/shared": "0.0.0"
|
|
34
|
+
},
|
|
35
|
+
"scripts": {
|
|
36
|
+
"dev": "tsx src/index.ts",
|
|
37
|
+
"build": "tsup",
|
|
38
|
+
"typecheck": "tsc -p tsconfig.json --noEmit",
|
|
39
|
+
"test": "vitest run --passWithNoTests",
|
|
40
|
+
"lint": "echo \"no lint yet\" && exit 0"
|
|
41
|
+
}
|
|
42
|
+
}
|