@plantnet/planttaxomatcher 0.1.2 → 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +864 -410
- package/package.json +42 -43
package/dist/index.js
CHANGED
|
@@ -1,6 +1,7 @@
|
|
|
1
1
|
#!/usr/bin/env node
|
|
2
2
|
|
|
3
3
|
// src/index.ts
|
|
4
|
+
import { createRequire } from "module";
|
|
4
5
|
import { promises as fs4 } from "fs";
|
|
5
6
|
import { parse as parsePath } from "path";
|
|
6
7
|
import { confirm, password } from "@inquirer/prompts";
|
|
@@ -31,7 +32,9 @@ var evidenceTypeSchema = z2.enum([
|
|
|
31
32
|
"local_fuzzy",
|
|
32
33
|
"external_fuzzy",
|
|
33
34
|
"team_history",
|
|
34
|
-
"llm"
|
|
35
|
+
"llm",
|
|
36
|
+
// Resolved via the OTHER backbone (WFO↔WCVP) then mapped back to the target.
|
|
37
|
+
"cross_backbone"
|
|
35
38
|
]);
|
|
36
39
|
var gradeSchema = z2.enum(["A", "B", "C"]);
|
|
37
40
|
var reviewStatusSchema = z2.enum(["not_required", "pending", "accepted", "rejected", "overridden"]);
|
|
@@ -59,8 +62,80 @@ var scoreBreakdownSchema = z2.object({
|
|
|
59
62
|
});
|
|
60
63
|
|
|
61
64
|
// ../shared/src/schemas/job.ts
|
|
65
|
+
import { z as z4 } from "zod";
|
|
66
|
+
|
|
67
|
+
// ../shared/src/schemas/plugins.ts
|
|
62
68
|
import { z as z3 } from "zod";
|
|
63
|
-
var
|
|
69
|
+
var verifierStateSchema = z3.enum(["pass", "fail", "error", "skipped"]);
|
|
70
|
+
var verifierColumnKindSchema = z3.enum(["check", "badge", "link", "text"]);
|
|
71
|
+
var verifierColumnSchema = z3.object({
|
|
72
|
+
/** Export column id, e.g. `verify_plantnet`. */
|
|
73
|
+
key: z3.string(),
|
|
74
|
+
/** Human-facing header. */
|
|
75
|
+
header: z3.string(),
|
|
76
|
+
kind: verifierColumnKindSchema
|
|
77
|
+
});
|
|
78
|
+
var pluginRunStatusSchema = z3.enum(["queued", "running", "completed", "failed", "cancelled"]);
|
|
79
|
+
var pluginRunTriggerSchema = z3.enum(["auto", "manual"]);
|
|
80
|
+
var pluginConfigOptionSchema = z3.object({
|
|
81
|
+
value: z3.string(),
|
|
82
|
+
label: z3.string()
|
|
83
|
+
});
|
|
84
|
+
var pluginConfigFieldSchema = z3.object({
|
|
85
|
+
key: z3.string(),
|
|
86
|
+
label: z3.string(),
|
|
87
|
+
type: z3.literal("select"),
|
|
88
|
+
options: z3.array(pluginConfigOptionSchema).min(1),
|
|
89
|
+
default: z3.string()
|
|
90
|
+
});
|
|
91
|
+
var pluginDescriptorSchema = z3.object({
|
|
92
|
+
id: z3.string(),
|
|
93
|
+
version: z3.string(),
|
|
94
|
+
title: z3.string(),
|
|
95
|
+
description: z3.string(),
|
|
96
|
+
/** Deployment can actually run it (key/env present). */
|
|
97
|
+
available: z3.boolean(),
|
|
98
|
+
columns: z3.array(verifierColumnSchema),
|
|
99
|
+
configFields: z3.array(pluginConfigFieldSchema)
|
|
100
|
+
});
|
|
101
|
+
var pluginRunRequestSchema = z3.object({
|
|
102
|
+
config: z3.record(z3.string(), z3.string()).default({})
|
|
103
|
+
});
|
|
104
|
+
var pluginRunSummarySchema = z3.object({
|
|
105
|
+
id: z3.string().uuid(),
|
|
106
|
+
jobId: z3.string().ulid(),
|
|
107
|
+
pluginId: z3.string(),
|
|
108
|
+
pluginVersion: z3.string(),
|
|
109
|
+
config: z3.record(z3.string(), z3.unknown()),
|
|
110
|
+
status: pluginRunStatusSchema,
|
|
111
|
+
trigger: pluginRunTriggerSchema,
|
|
112
|
+
totalUnits: z3.number().int().nonnegative(),
|
|
113
|
+
processedUnits: z3.number().int().nonnegative(),
|
|
114
|
+
passUnits: z3.number().int().nonnegative(),
|
|
115
|
+
errorUnits: z3.number().int().nonnegative(),
|
|
116
|
+
/** Label of the token that triggered a manual run; null for auto runs. */
|
|
117
|
+
triggeredBy: z3.string().nullable(),
|
|
118
|
+
/** `failed` run: why it stopped. `completed` run: why its errored rows
|
|
119
|
+
* couldn't be checked (most frequent reasons first). */
|
|
120
|
+
error: z3.string().nullable(),
|
|
121
|
+
createdAt: z3.string().datetime(),
|
|
122
|
+
startedAt: z3.string().datetime().nullable(),
|
|
123
|
+
completedAt: z3.string().datetime().nullable()
|
|
124
|
+
});
|
|
125
|
+
var pluginRunsResponseSchema = z3.object({
|
|
126
|
+
runs: z3.array(pluginRunSummarySchema)
|
|
127
|
+
});
|
|
128
|
+
var jobRowPluginResultSchema = z3.object({
|
|
129
|
+
pluginId: z3.string(),
|
|
130
|
+
state: verifierStateSchema,
|
|
131
|
+
label: z3.string().nullable(),
|
|
132
|
+
url: z3.string().nullable(),
|
|
133
|
+
/** Why this verdict — the error reason, or what a `fail` was checked against. */
|
|
134
|
+
detail: z3.string().nullable()
|
|
135
|
+
});
|
|
136
|
+
|
|
137
|
+
// ../shared/src/schemas/job.ts
|
|
138
|
+
var jobStatusSchema = z4.enum([
|
|
64
139
|
"queued",
|
|
65
140
|
"parsing",
|
|
66
141
|
"matching",
|
|
@@ -69,39 +144,87 @@ var jobStatusSchema = z3.enum([
|
|
|
69
144
|
"completed",
|
|
70
145
|
"failed"
|
|
71
146
|
]);
|
|
72
|
-
var authorModeSchema =
|
|
73
|
-
var reviewModeSchema =
|
|
74
|
-
var jobConfigSchema =
|
|
75
|
-
nameColumn:
|
|
76
|
-
idColumn:
|
|
77
|
-
familyColumn:
|
|
78
|
-
genusColumn:
|
|
79
|
-
rankColumn:
|
|
80
|
-
authorColumn:
|
|
81
|
-
sourceReferentialColumn:
|
|
82
|
-
sourceIdColumn:
|
|
147
|
+
var authorModeSchema = z4.enum(["ignore", "prefer", "strict"]);
|
|
148
|
+
var reviewModeSchema = z4.enum(["off", "recommended", "strict"]);
|
|
149
|
+
var jobConfigSchema = z4.object({
|
|
150
|
+
nameColumn: z4.string().min(1),
|
|
151
|
+
idColumn: z4.string().nullable().optional(),
|
|
152
|
+
familyColumn: z4.string().nullable().optional(),
|
|
153
|
+
genusColumn: z4.string().nullable().optional(),
|
|
154
|
+
rankColumn: z4.string().nullable().optional(),
|
|
155
|
+
authorColumn: z4.string().nullable().optional(),
|
|
156
|
+
sourceReferentialColumn: z4.string().nullable().optional(),
|
|
157
|
+
sourceIdColumn: z4.string().nullable().optional(),
|
|
83
158
|
authorMode: authorModeSchema.default("prefer"),
|
|
84
|
-
matchAuthors:
|
|
85
|
-
parallelism:
|
|
86
|
-
allowFuzzy:
|
|
87
|
-
allowLlm:
|
|
88
|
-
llmCostCapCents:
|
|
159
|
+
matchAuthors: z4.boolean().default(true),
|
|
160
|
+
parallelism: z4.number().int().min(1).max(10).default(4),
|
|
161
|
+
allowFuzzy: z4.boolean().default(true),
|
|
162
|
+
allowLlm: z4.boolean().default(false),
|
|
163
|
+
llmCostCapCents: z4.number().int().min(0).default(500),
|
|
89
164
|
reviewMode: reviewModeSchema.default("recommended"),
|
|
90
|
-
exportConfirmedOnly:
|
|
165
|
+
exportConfirmedOnly: z4.boolean().default(false),
|
|
166
|
+
/**
|
|
167
|
+
* When false, this job's review decisions (accept + reject) do NOT feed the
|
|
168
|
+
* team's Layer 0.5 match-history cache — a one-off or experimental job can't
|
|
169
|
+
* teach (or poison) the shared cache. Default false (opt-in).
|
|
170
|
+
*/
|
|
171
|
+
contributeToTeamCache: z4.boolean().default(false),
|
|
172
|
+
/**
|
|
173
|
+
* Roll infraspecific results up to the species: when an input resolves to an
|
|
174
|
+
* infraspecific accepted taxon (Variety / Subspecies / Form / …), replace it
|
|
175
|
+
* with its parent Species so every output row sits at species level.
|
|
176
|
+
*
|
|
177
|
+
* Default FALSE — the resolved rank is preserved as-is. Rolling up discards
|
|
178
|
+
* information the source data carried, so it is opt-in: turn it on only when
|
|
179
|
+
* the consuming system works at species level and you would otherwise have to
|
|
180
|
+
* flatten the export yourself.
|
|
181
|
+
*/
|
|
182
|
+
speciesLevelAcceptedOnly: z4.boolean().default(false),
|
|
183
|
+
/**
|
|
184
|
+
* "The author isn't important." When the canonical name matches exactly one
|
|
185
|
+
* taxon but the input's author could not be CONFIRMED (flag
|
|
186
|
+
* `author-unconfirmed`), auto-accept the match instead of sending it to
|
|
187
|
+
* review. The taxon itself was never in doubt in that case — only whether
|
|
188
|
+
* the author string cites it the way WCVP does — so a dataset whose author
|
|
189
|
+
* column is unreliable (or absent from the source) can skip that queue.
|
|
190
|
+
*
|
|
191
|
+
* Default FALSE. This does NOT relax a genuine author CONFLICT
|
|
192
|
+
* (`author-mismatch`): a conflicting author may point at a different plant,
|
|
193
|
+
* so those still go to review. Every other review trigger (qualifiers,
|
|
194
|
+
* parse quality, force-review families, alternatives) is untouched.
|
|
195
|
+
*/
|
|
196
|
+
acceptUnconfirmedAuthor: z4.boolean().default(false),
|
|
197
|
+
/**
|
|
198
|
+
* Which taxonomic backbone to match against. `wcvp` (default) runs the full
|
|
199
|
+
* cascade; `wfo` matches against the World Flora Online snapshot using the
|
|
200
|
+
* local layers (L1–L4). Stored on `jobs.referential`.
|
|
201
|
+
*/
|
|
202
|
+
referential: z4.enum(["wcvp", "wfo"]).default("wcvp"),
|
|
203
|
+
/**
|
|
204
|
+
* Which snapshot version of the chosen backbone to match against. When
|
|
205
|
+
* omitted, the API resolves it at submit time: the team's configured default
|
|
206
|
+
* snapshot for this backbone (Admin → Backbone) when it is still installed,
|
|
207
|
+
* otherwise the most-recently-imported one. The job stores the resolved
|
|
208
|
+
* version on `jobs.referential_version` so re-runs are reproducible even if
|
|
209
|
+
* a newer snapshot lands later.
|
|
210
|
+
*/
|
|
211
|
+
referentialVersion: z4.string().min(1).optional(),
|
|
91
212
|
/**
|
|
92
|
-
*
|
|
93
|
-
*
|
|
94
|
-
*
|
|
95
|
-
*
|
|
213
|
+
* WGSRPD Level-3 area code (e.g. 'MAS' = Massachusetts) the import is scoped
|
|
214
|
+
* to. When set, an ambiguous multi-candidate match is narrowed to the taxa
|
|
215
|
+
* that occur in this area — exactly one survivor resolves the match (Grade B,
|
|
216
|
+
* flagged `resolved-by-area:<code>`). null/omitted = no disambiguation.
|
|
96
217
|
*/
|
|
97
|
-
|
|
218
|
+
area: z4.string().nullable().optional(),
|
|
98
219
|
/**
|
|
99
|
-
*
|
|
100
|
-
*
|
|
101
|
-
*
|
|
102
|
-
*
|
|
220
|
+
* Auto-run the Pl@ntNet verification plugin for this job at completion —
|
|
221
|
+
* tags each match with whether its accepted taxon aligns to a Pl@ntNet
|
|
222
|
+
* species. Defaults on; set false to skip the auto-run for this job (it can
|
|
223
|
+
* still be triggered manually from the Plugins menu). Only has an effect
|
|
224
|
+
* when the server has Pl@ntNet configured (key set + not globally disabled);
|
|
225
|
+
* otherwise it is a no-op regardless.
|
|
103
226
|
*/
|
|
104
|
-
|
|
227
|
+
plantnet: z4.boolean().optional(),
|
|
105
228
|
/**
|
|
106
229
|
* Row filter: keep only rows whose `filterColumn` value equals
|
|
107
230
|
* `filterValue` (trimmed, case-insensitive); all other rows are skipped at
|
|
@@ -109,306 +232,403 @@ var jobConfigSchema = z3.object({
|
|
|
109
232
|
* apply. Lets a user match a subset of a mixed file (e.g. only
|
|
110
233
|
* `kingdom = Plantae`).
|
|
111
234
|
*/
|
|
112
|
-
filterColumn:
|
|
113
|
-
filterValue:
|
|
235
|
+
filterColumn: z4.string().nullable().optional(),
|
|
236
|
+
filterValue: z4.string().nullable().optional()
|
|
114
237
|
});
|
|
115
|
-
var wcvpSnapshotSummarySchema =
|
|
116
|
-
version:
|
|
117
|
-
recordCount:
|
|
118
|
-
importedAt:
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
|
|
126
|
-
|
|
127
|
-
|
|
128
|
-
|
|
129
|
-
|
|
130
|
-
|
|
238
|
+
var wcvpSnapshotSummarySchema = z4.object({
|
|
239
|
+
version: z4.string(),
|
|
240
|
+
recordCount: z4.number().int().nonnegative(),
|
|
241
|
+
importedAt: z4.string().datetime(),
|
|
242
|
+
/** Newest OFFICIAL snapshot (derived ones never count as latest). */
|
|
243
|
+
isLatest: z4.boolean(),
|
|
244
|
+
kind: z4.enum(["official", "derived"]),
|
|
245
|
+
baseVersion: z4.string().nullable(),
|
|
246
|
+
label: z4.string().nullable(),
|
|
247
|
+
ownerTeamId: z4.string().nullable()
|
|
248
|
+
});
|
|
249
|
+
var idTypeDetectionSchema = z4.object({
|
|
250
|
+
sampleSize: z4.number().int().nonnegative(),
|
|
251
|
+
counts: z4.record(z4.string(), z4.number().int().nonnegative()),
|
|
252
|
+
dominant: z4.string().nullable(),
|
|
253
|
+
dominantConfidence: z4.number().min(0).max(1),
|
|
254
|
+
minorityExamples: z4.array(
|
|
255
|
+
z4.object({
|
|
256
|
+
rowIndex: z4.number().int().nonnegative(),
|
|
257
|
+
value: z4.string(),
|
|
258
|
+
type: z4.string()
|
|
131
259
|
})
|
|
132
260
|
)
|
|
133
261
|
});
|
|
134
|
-
var publicAccessSchema =
|
|
135
|
-
var jobSummarySchema =
|
|
136
|
-
id:
|
|
137
|
-
teamId:
|
|
138
|
-
userId:
|
|
262
|
+
var publicAccessSchema = z4.enum(["none", "read", "review"]);
|
|
263
|
+
var jobSummarySchema = z4.object({
|
|
264
|
+
id: z4.string().ulid(),
|
|
265
|
+
teamId: z4.string().uuid(),
|
|
266
|
+
userId: z4.string().uuid(),
|
|
139
267
|
/** Optional user-supplied label. URL still uses `id`; null when unset. */
|
|
140
|
-
name:
|
|
141
|
-
/**
|
|
142
|
-
*
|
|
143
|
-
|
|
268
|
+
name: z4.string().nullable(),
|
|
269
|
+
/** The job's author: the label of the token that submitted it (snapshotted
|
|
270
|
+
* at creation, so it survives the token being renamed or revoked). Null for
|
|
271
|
+
* jobs created before authorship was tracked. */
|
|
272
|
+
submittedBy: z4.string().nullable(),
|
|
144
273
|
status: jobStatusSchema,
|
|
145
274
|
/** 'none' = private. 'read'/'review' = anyone with the link, no sign-in. */
|
|
146
275
|
publicAccess: publicAccessSchema,
|
|
147
|
-
referentialVersion:
|
|
148
|
-
idColumn:
|
|
276
|
+
referentialVersion: z4.string(),
|
|
277
|
+
idColumn: z4.string().nullable(),
|
|
149
278
|
idTypeDetection: idTypeDetectionSchema.nullable(),
|
|
150
|
-
totalRows:
|
|
151
|
-
uniqueQueries:
|
|
152
|
-
processedRows:
|
|
153
|
-
matchedRows:
|
|
154
|
-
ambiguousRows:
|
|
155
|
-
errorRows:
|
|
156
|
-
needsReviewRows:
|
|
157
|
-
llmCostCents:
|
|
158
|
-
llmCostCapCents:
|
|
159
|
-
createdAt:
|
|
160
|
-
startedAt:
|
|
161
|
-
completedAt:
|
|
162
|
-
expiresAt:
|
|
279
|
+
totalRows: z4.number().int().nonnegative(),
|
|
280
|
+
uniqueQueries: z4.number().int().nonnegative(),
|
|
281
|
+
processedRows: z4.number().int().nonnegative(),
|
|
282
|
+
matchedRows: z4.number().int().nonnegative(),
|
|
283
|
+
ambiguousRows: z4.number().int().nonnegative(),
|
|
284
|
+
errorRows: z4.number().int().nonnegative(),
|
|
285
|
+
needsReviewRows: z4.number().int().nonnegative(),
|
|
286
|
+
llmCostCents: z4.number().int().nonnegative(),
|
|
287
|
+
llmCostCapCents: z4.number().int().nonnegative(),
|
|
288
|
+
createdAt: z4.string().datetime(),
|
|
289
|
+
startedAt: z4.string().datetime().nullable(),
|
|
290
|
+
completedAt: z4.string().datetime().nullable(),
|
|
291
|
+
expiresAt: z4.string().datetime()
|
|
163
292
|
});
|
|
164
|
-
var jobPublicAccessUpdateSchema =
|
|
293
|
+
var jobPublicAccessUpdateSchema = z4.object({
|
|
165
294
|
publicAccess: publicAccessSchema
|
|
166
295
|
});
|
|
167
|
-
var
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
296
|
+
var jobRenameSchema = z4.object({
|
|
297
|
+
name: z4.string().trim().min(1).max(200)
|
|
298
|
+
});
|
|
299
|
+
var progressEventSchema = z4.object({
|
|
300
|
+
type: z4.enum(["progress", "status", "error", "completed"]),
|
|
301
|
+
jobId: z4.string().ulid(),
|
|
302
|
+
timestamp: z4.string().datetime(),
|
|
303
|
+
processedRows: z4.number().int().nonnegative().optional(),
|
|
304
|
+
totalRows: z4.number().int().nonnegative().optional(),
|
|
173
305
|
status: jobStatusSchema.optional(),
|
|
174
|
-
message:
|
|
306
|
+
message: z4.string().optional()
|
|
175
307
|
});
|
|
176
|
-
var taxonIdentifierRefSchema =
|
|
177
|
-
namespace:
|
|
178
|
-
value:
|
|
308
|
+
var taxonIdentifierRefSchema = z4.object({
|
|
309
|
+
namespace: z4.string(),
|
|
310
|
+
value: z4.string()
|
|
179
311
|
});
|
|
180
|
-
var jobRowSummarySchema =
|
|
181
|
-
id:
|
|
182
|
-
rowIndex:
|
|
183
|
-
inputName:
|
|
184
|
-
inputId:
|
|
185
|
-
inputFamily:
|
|
312
|
+
var jobRowSummarySchema = z4.object({
|
|
313
|
+
id: z4.string().uuid(),
|
|
314
|
+
rowIndex: z4.number().int().nonnegative(),
|
|
315
|
+
inputName: z4.string().nullable(),
|
|
316
|
+
inputId: z4.string().nullable(),
|
|
317
|
+
inputFamily: z4.string().nullable(),
|
|
186
318
|
matchStatus: matchStatusSchema.nullable(),
|
|
187
319
|
grade: gradeSchema.nullable(),
|
|
188
|
-
evidenceType:
|
|
320
|
+
evidenceType: z4.string().nullable(),
|
|
189
321
|
reviewStatus: reviewStatusSchema,
|
|
190
|
-
matchQueryId:
|
|
191
|
-
confidence:
|
|
192
|
-
layer:
|
|
193
|
-
flags:
|
|
194
|
-
candidateAcceptedName:
|
|
322
|
+
matchQueryId: z4.string().uuid().nullable(),
|
|
323
|
+
confidence: z4.number().nullable(),
|
|
324
|
+
layer: z4.string().nullable(),
|
|
325
|
+
flags: z4.array(z4.string()).nullable(),
|
|
326
|
+
candidateAcceptedName: z4.string().nullable(),
|
|
195
327
|
/** Authorship of the accepted taxon (distinct from `candidateAuthorship`
|
|
196
328
|
* which carries the matched-row author when the match resolves via a synonym). */
|
|
197
|
-
candidateAcceptedAuthorship:
|
|
198
|
-
candidateAcceptedIdentifiers:
|
|
199
|
-
candidateScientificName:
|
|
200
|
-
candidateAuthorship:
|
|
201
|
-
candidateFamily:
|
|
202
|
-
candidateReason:
|
|
203
|
-
candidateTargetUrl:
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
329
|
+
candidateAcceptedAuthorship: z4.string().nullable(),
|
|
330
|
+
candidateAcceptedIdentifiers: z4.array(taxonIdentifierRefSchema).nullable(),
|
|
331
|
+
candidateScientificName: z4.string().nullable(),
|
|
332
|
+
candidateAuthorship: z4.string().nullable(),
|
|
333
|
+
candidateFamily: z4.string().nullable(),
|
|
334
|
+
candidateReason: z4.string().nullable(),
|
|
335
|
+
candidateTargetUrl: z4.string().nullable(),
|
|
336
|
+
/** Post-job verification-plugin results for this row, one per plugin that
|
|
337
|
+
* has a completed run. Empty when no plugin has run. Sourced from
|
|
338
|
+
* `job_plugin_results` joined via the row's match query. */
|
|
339
|
+
plugins: z4.array(jobRowPluginResultSchema).default([])
|
|
340
|
+
});
|
|
341
|
+
var gradeCountsSchema = z4.object({
|
|
342
|
+
A: z4.number().int().nonnegative(),
|
|
343
|
+
B: z4.number().int().nonnegative(),
|
|
344
|
+
C: z4.number().int().nonnegative(),
|
|
345
|
+
ungraded: z4.number().int().nonnegative()
|
|
346
|
+
});
|
|
347
|
+
var statusCountsSchema = z4.object({
|
|
348
|
+
matched: z4.number().int().nonnegative(),
|
|
349
|
+
ambiguous: z4.number().int().nonnegative(),
|
|
350
|
+
no_match: z4.number().int().nonnegative(),
|
|
351
|
+
error: z4.number().int().nonnegative(),
|
|
352
|
+
skipped: z4.number().int().nonnegative(),
|
|
217
353
|
/** Rows whose match query exists but hasn't been processed yet (mid-run).
|
|
218
354
|
* Distinct from `skipped` (no query — empty/guarded input). Not part of the
|
|
219
355
|
* outcome funnel; the SPA can show it as in-progress. */
|
|
220
|
-
pending:
|
|
356
|
+
pending: z4.number().int().nonnegative()
|
|
221
357
|
});
|
|
222
|
-
var
|
|
223
|
-
|
|
358
|
+
var layerCountsSchema = z4.record(z4.string(), z4.number().int().nonnegative());
|
|
359
|
+
var jobRowsPageSchema = z4.object({
|
|
360
|
+
rows: z4.array(jobRowSummarySchema),
|
|
224
361
|
/** Row count matching the CURRENT filter (not the whole job) — drives
|
|
225
362
|
* pagination so page count adapts to active grade/status/search/conf filters. */
|
|
226
|
-
total:
|
|
227
|
-
offset:
|
|
228
|
-
limit:
|
|
363
|
+
total: z4.number().int().nonnegative(),
|
|
364
|
+
offset: z4.number().int().nonnegative(),
|
|
365
|
+
limit: z4.number().int().positive(),
|
|
229
366
|
/** Per-grade row counts for the whole job, independent of the current
|
|
230
367
|
* filter. Used by the SPA to show "B (12)" next to each grade chip. */
|
|
231
368
|
gradeCounts: gradeCountsSchema,
|
|
232
369
|
/** Per-status row counts (matched / ambiguous / no_match / error /
|
|
233
370
|
* skipped), same scope + purpose as `gradeCounts`. */
|
|
234
|
-
statusCounts: statusCountsSchema
|
|
235
|
-
|
|
236
|
-
|
|
237
|
-
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
|
|
241
|
-
|
|
242
|
-
|
|
243
|
-
|
|
244
|
-
|
|
245
|
-
|
|
246
|
-
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
|
|
250
|
-
|
|
251
|
-
|
|
252
|
-
|
|
253
|
-
|
|
254
|
-
|
|
255
|
-
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
260
|
-
|
|
261
|
-
|
|
371
|
+
statusCounts: statusCountsSchema,
|
|
372
|
+
/** Per-layer row counts, keyed by evidence type. Same scope + purpose as
|
|
373
|
+
* `gradeCounts`; only layers present in the job appear. */
|
|
374
|
+
layerCounts: layerCountsSchema
|
|
375
|
+
});
|
|
376
|
+
var wcvpRankLevelSchema = z4.enum(["genus", "species", "infra"]);
|
|
377
|
+
var wcvpSearchHitSchema = z4.object({
|
|
378
|
+
taxonId: z4.string(),
|
|
379
|
+
scientificName: z4.string(),
|
|
380
|
+
canonicalName: z4.string(),
|
|
381
|
+
authorship: z4.string().nullable(),
|
|
382
|
+
rank: z4.string().nullable(),
|
|
383
|
+
taxonomicStatus: z4.string().nullable(),
|
|
384
|
+
family: z4.string().nullable(),
|
|
385
|
+
acceptedTaxonId: z4.string().nullable(),
|
|
386
|
+
acceptedName: z4.string().nullable(),
|
|
387
|
+
similarity: z4.number()
|
|
388
|
+
});
|
|
389
|
+
var wcvpSearchResponseSchema = z4.array(wcvpSearchHitSchema);
|
|
390
|
+
var speciesRollupStatusSchema = z4.enum([
|
|
391
|
+
"available",
|
|
392
|
+
"already-species",
|
|
393
|
+
"no-species-ancestor",
|
|
394
|
+
"no-selection",
|
|
395
|
+
"unknown-taxon"
|
|
396
|
+
]);
|
|
397
|
+
var speciesRollupTaxonSchema = z4.object({
|
|
398
|
+
taxonId: z4.string(),
|
|
399
|
+
scientificName: z4.string(),
|
|
400
|
+
canonicalName: z4.string(),
|
|
401
|
+
authorship: z4.string().nullable(),
|
|
402
|
+
rank: z4.string().nullable(),
|
|
403
|
+
family: z4.string().nullable()
|
|
404
|
+
});
|
|
405
|
+
var speciesRollupResponseSchema = z4.object({
|
|
406
|
+
status: speciesRollupStatusSchema,
|
|
407
|
+
/** The accepted taxon the row currently resolves to. */
|
|
408
|
+
current: speciesRollupTaxonSchema.nullable(),
|
|
409
|
+
/** Only set when `status` is `available`. */
|
|
410
|
+
target: speciesRollupTaxonSchema.nullable()
|
|
411
|
+
});
|
|
412
|
+
var jobDiffRowSchema = z4.object({
|
|
413
|
+
rowId: z4.string().uuid(),
|
|
414
|
+
rowIndex: z4.number().int().nonnegative(),
|
|
415
|
+
inputName: z4.string().nullable(),
|
|
416
|
+
inputFamily: z4.string().nullable(),
|
|
417
|
+
matchedName: z4.string().nullable(),
|
|
418
|
+
matchedAuthorship: z4.string().nullable(),
|
|
419
|
+
matchedTaxonomicStatus: z4.string().nullable(),
|
|
420
|
+
acceptedName: z4.string().nullable(),
|
|
421
|
+
acceptedAuthorship: z4.string().nullable(),
|
|
422
|
+
acceptedFamily: z4.string().nullable(),
|
|
423
|
+
isSynonym: z4.boolean(),
|
|
424
|
+
familyChanged: z4.boolean(),
|
|
262
425
|
grade: gradeSchema.nullable(),
|
|
263
426
|
reviewStatus: reviewStatusSchema,
|
|
264
|
-
targetUrl:
|
|
427
|
+
targetUrl: z4.string().nullable()
|
|
265
428
|
});
|
|
266
|
-
var jobDiffPageSchema =
|
|
267
|
-
rows:
|
|
268
|
-
total:
|
|
429
|
+
var jobDiffPageSchema = z4.object({
|
|
430
|
+
rows: z4.array(jobDiffRowSchema),
|
|
431
|
+
total: z4.number().int().nonnegative()
|
|
269
432
|
});
|
|
270
|
-
var reviewActionSchema =
|
|
271
|
-
var reviewRequestSchema =
|
|
433
|
+
var reviewActionSchema = z4.enum(["accept", "reject"]);
|
|
434
|
+
var reviewRequestSchema = z4.object({
|
|
272
435
|
action: reviewActionSchema,
|
|
273
|
-
|
|
436
|
+
// Optional so a row with no candidates can still be resolved as "no
|
|
437
|
+
// match" (reject with no candidate). Accept always needs one.
|
|
438
|
+
candidateId: z4.string().uuid().optional()
|
|
439
|
+
}).refine((v) => v.action === "reject" || !!v.candidateId, {
|
|
440
|
+
message: "candidateId is required to accept a candidate",
|
|
441
|
+
path: ["candidateId"]
|
|
274
442
|
});
|
|
275
|
-
var bulkReviewRequestSchema =
|
|
443
|
+
var bulkReviewRequestSchema = z4.object({
|
|
276
444
|
action: reviewActionSchema,
|
|
277
|
-
filter:
|
|
445
|
+
filter: z4.object({
|
|
278
446
|
grade: gradeSchema.optional(),
|
|
279
447
|
matchStatus: matchStatusSchema.optional(),
|
|
280
|
-
family:
|
|
448
|
+
family: z4.string().optional(),
|
|
449
|
+
/** Restrict the bulk action to these specific match-query ids. The
|
|
450
|
+
* review queue's grouped/batch view computes a group's membership
|
|
451
|
+
* client-side (by issue category or author-equivalence pair) and
|
|
452
|
+
* sends the exact ids — flag/category logic that a coarse
|
|
453
|
+
* grade/family filter can't express. ANDed with any other filter;
|
|
454
|
+
* the server still restricts to currently-pending, owned queries. */
|
|
455
|
+
matchQueryIds: z4.array(z4.string().uuid()).max(5e4).optional()
|
|
281
456
|
}).default({}),
|
|
282
|
-
dryRun:
|
|
457
|
+
dryRun: z4.boolean().default(false)
|
|
283
458
|
});
|
|
284
|
-
var bulkReviewResponseSchema =
|
|
459
|
+
var bulkReviewResponseSchema = z4.object({
|
|
285
460
|
action: reviewActionSchema,
|
|
286
|
-
matchedQueries:
|
|
287
|
-
dryRun:
|
|
288
|
-
appliedAt:
|
|
461
|
+
matchedQueries: z4.number().int().nonnegative(),
|
|
462
|
+
dryRun: z4.boolean(),
|
|
463
|
+
appliedAt: z4.string().datetime().nullable()
|
|
289
464
|
});
|
|
290
|
-
var reviewResponseSchema =
|
|
291
|
-
matchQueryId:
|
|
465
|
+
var reviewResponseSchema = z4.object({
|
|
466
|
+
matchQueryId: z4.string().uuid(),
|
|
292
467
|
action: reviewActionSchema,
|
|
293
|
-
|
|
468
|
+
// Null only for a "no match" reject (a row with no candidate). An accept
|
|
469
|
+
// always resolves to a candidate — the refine catches a server that
|
|
470
|
+
// returned an accept without one.
|
|
471
|
+
candidateId: z4.string().uuid().nullable(),
|
|
294
472
|
reviewStatus: reviewStatusSchema,
|
|
295
|
-
reviewEventId:
|
|
473
|
+
reviewEventId: z4.string().uuid()
|
|
474
|
+
}).refine((v) => v.action !== "accept" || v.candidateId !== null, {
|
|
475
|
+
message: "an accept must resolve to a candidateId",
|
|
476
|
+
path: ["candidateId"]
|
|
296
477
|
});
|
|
297
|
-
var matchAttemptSummarySchema =
|
|
298
|
-
id:
|
|
299
|
-
layer:
|
|
300
|
-
provider:
|
|
301
|
-
status:
|
|
302
|
-
durationMs:
|
|
478
|
+
var matchAttemptSummarySchema = z4.object({
|
|
479
|
+
id: z4.string().uuid(),
|
|
480
|
+
layer: z4.string(),
|
|
481
|
+
provider: z4.string(),
|
|
482
|
+
status: z4.string(),
|
|
483
|
+
durationMs: z4.number().int().nonnegative(),
|
|
303
484
|
// JSONB fields accept any shape. Note: `z.unknown()` infers as optional,
|
|
304
485
|
// so consumers should guard for `undefined` even though we always send
|
|
305
486
|
// them as part of the response.
|
|
306
|
-
query:
|
|
307
|
-
rawResponseSummary:
|
|
308
|
-
createdAt:
|
|
487
|
+
query: z4.unknown(),
|
|
488
|
+
rawResponseSummary: z4.unknown().nullable(),
|
|
489
|
+
createdAt: z4.string().datetime()
|
|
309
490
|
});
|
|
310
|
-
var matchCandidateSummarySchema =
|
|
311
|
-
id:
|
|
312
|
-
attemptId:
|
|
313
|
-
source:
|
|
314
|
-
sourceId:
|
|
315
|
-
scientificName:
|
|
316
|
-
canonicalName:
|
|
317
|
-
authorship:
|
|
318
|
-
rank:
|
|
319
|
-
taxonomicStatus:
|
|
320
|
-
acceptedName:
|
|
321
|
-
acceptedAuthorship:
|
|
322
|
-
acceptedIdentifiers:
|
|
323
|
-
identifiers:
|
|
324
|
-
family:
|
|
325
|
-
confidence:
|
|
491
|
+
var matchCandidateSummarySchema = z4.object({
|
|
492
|
+
id: z4.string().uuid(),
|
|
493
|
+
attemptId: z4.string().uuid(),
|
|
494
|
+
source: z4.string(),
|
|
495
|
+
sourceId: z4.string().nullable(),
|
|
496
|
+
scientificName: z4.string(),
|
|
497
|
+
canonicalName: z4.string(),
|
|
498
|
+
authorship: z4.string().nullable(),
|
|
499
|
+
rank: z4.string().nullable(),
|
|
500
|
+
taxonomicStatus: z4.string().nullable(),
|
|
501
|
+
acceptedName: z4.string().nullable(),
|
|
502
|
+
acceptedAuthorship: z4.string().nullable(),
|
|
503
|
+
acceptedIdentifiers: z4.array(taxonIdentifierRefSchema).nullable(),
|
|
504
|
+
identifiers: z4.array(taxonIdentifierRefSchema),
|
|
505
|
+
family: z4.string().nullable(),
|
|
506
|
+
confidence: z4.number().nullable(),
|
|
326
507
|
grade: gradeSchema.nullable(),
|
|
327
|
-
reason:
|
|
328
|
-
flags:
|
|
329
|
-
state:
|
|
330
|
-
targetUrl:
|
|
508
|
+
reason: z4.string().nullable(),
|
|
509
|
+
flags: z4.array(z4.string()).nullable(),
|
|
510
|
+
state: z4.enum(["proposed", "accepted", "rejected", "overridden"]),
|
|
511
|
+
targetUrl: z4.string().nullable()
|
|
331
512
|
});
|
|
332
|
-
var matchQueryDetailSchema =
|
|
333
|
-
id:
|
|
334
|
-
jobId:
|
|
335
|
-
normalizedInput:
|
|
336
|
-
parsed:
|
|
513
|
+
var matchQueryDetailSchema = z4.object({
|
|
514
|
+
id: z4.string().uuid(),
|
|
515
|
+
jobId: z4.string().ulid(),
|
|
516
|
+
normalizedInput: z4.string(),
|
|
517
|
+
parsed: z4.unknown(),
|
|
337
518
|
matchStatus: matchStatusSchema.nullable(),
|
|
338
|
-
evidenceType:
|
|
519
|
+
evidenceType: z4.string().nullable(),
|
|
339
520
|
grade: gradeSchema.nullable(),
|
|
340
|
-
confidence:
|
|
341
|
-
flags:
|
|
521
|
+
confidence: z4.number().nullable(),
|
|
522
|
+
flags: z4.array(z4.string()).nullable(),
|
|
342
523
|
reviewStatus: reviewStatusSchema,
|
|
343
|
-
selectedCandidateId:
|
|
524
|
+
selectedCandidateId: z4.string().uuid().nullable()
|
|
525
|
+
});
|
|
526
|
+
var otherVersionMatchSchema = z4.object({
|
|
527
|
+
version: z4.string(),
|
|
528
|
+
importedAt: z4.string(),
|
|
529
|
+
isLatest: z4.boolean(),
|
|
530
|
+
scientificName: z4.string(),
|
|
531
|
+
authorship: z4.string().nullable(),
|
|
532
|
+
taxonomicStatus: z4.string().nullable(),
|
|
533
|
+
acceptedName: z4.string().nullable()
|
|
344
534
|
});
|
|
345
|
-
var queryCandidatesResponseSchema =
|
|
535
|
+
var queryCandidatesResponseSchema = z4.object({
|
|
346
536
|
query: matchQueryDetailSchema,
|
|
347
|
-
candidates:
|
|
348
|
-
attempts:
|
|
349
|
-
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
|
|
353
|
-
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
|
|
365
|
-
|
|
366
|
-
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
|
|
370
|
-
|
|
371
|
-
|
|
537
|
+
candidates: z4.array(matchCandidateSummarySchema),
|
|
538
|
+
attempts: z4.array(matchAttemptSummarySchema),
|
|
539
|
+
/** A newer/other backbone version that has this exact name (see schema). */
|
|
540
|
+
otherVersionMatch: otherVersionMatchSchema.nullable()
|
|
541
|
+
});
|
|
542
|
+
var reviewChatMessageSchema = z4.object({
|
|
543
|
+
role: z4.enum(["user", "assistant"]),
|
|
544
|
+
content: z4.string().min(1).max(4e3)
|
|
545
|
+
});
|
|
546
|
+
var reviewChatBodySchema = z4.object({
|
|
547
|
+
messages: z4.array(reviewChatMessageSchema).min(1).max(40),
|
|
548
|
+
/** Let the assistant also search the web (OpenRouter web plugin). Default on
|
|
549
|
+
* server-side when omitted; the client sends it explicitly from its toggle. */
|
|
550
|
+
webSearch: z4.boolean().optional(),
|
|
551
|
+
/** Per-chat model override (an OpenRouter id / alias). When omitted the
|
|
552
|
+
* team's configured heavy model is used. Length-capped defensively. */
|
|
553
|
+
model: z4.string().trim().min(1).max(200).optional()
|
|
554
|
+
});
|
|
555
|
+
var reviewChatResponseSchema = z4.object({
|
|
556
|
+
reply: z4.string(),
|
|
557
|
+
model: z4.string()
|
|
558
|
+
});
|
|
559
|
+
var reviewChatStreamEventSchema = z4.discriminatedUnion("type", [
|
|
560
|
+
/** A chunk of the model's visible answer. */
|
|
561
|
+
z4.object({ type: z4.literal("text"), delta: z4.string() }),
|
|
562
|
+
/** A chunk of the model's reasoning/thinking (when the model exposes it). */
|
|
563
|
+
z4.object({ type: z4.literal("reasoning"), delta: z4.string() }),
|
|
564
|
+
/** The model decided to call a tool, with the (JSON) arguments it chose. */
|
|
565
|
+
z4.object({ type: z4.literal("tool_call"), id: z4.string(), name: z4.string(), arguments: z4.string() }),
|
|
566
|
+
/** The result we fed back to the model after running that tool. `sql` is the
|
|
567
|
+
* statement the tool actually ran, shown to the reviewer for transparency —
|
|
568
|
+
* it is emitted on this event ONLY and never added to the model's context. */
|
|
569
|
+
z4.object({
|
|
570
|
+
type: z4.literal("tool_result"),
|
|
571
|
+
id: z4.string(),
|
|
572
|
+
name: z4.string(),
|
|
573
|
+
result: z4.string(),
|
|
574
|
+
sql: z4.string().optional()
|
|
575
|
+
}),
|
|
576
|
+
/** Terminal success — carries the model id that answered. */
|
|
577
|
+
z4.object({ type: z4.literal("done"), model: z4.string() }),
|
|
578
|
+
/** Terminal failure — a human-readable reason. */
|
|
579
|
+
z4.object({ type: z4.literal("error"), error: z4.string() })
|
|
580
|
+
]);
|
|
581
|
+
var replayableLayerSchema = z4.enum(["L1", "L2", "L3", "L4", "L5", "L6", "L7"]);
|
|
582
|
+
var replayLayerBodySchema = z4.object({ layer: replayableLayerSchema });
|
|
583
|
+
var replayCandidateSchema = z4.object({
|
|
584
|
+
scientificName: z4.string(),
|
|
585
|
+
canonicalName: z4.string(),
|
|
586
|
+
authorship: z4.string().nullable(),
|
|
587
|
+
rank: z4.string().nullable(),
|
|
588
|
+
taxonomicStatus: z4.string().nullable(),
|
|
589
|
+
acceptedName: z4.string().nullable(),
|
|
590
|
+
acceptedAuthorship: z4.string().nullable(),
|
|
591
|
+
family: z4.string().nullable(),
|
|
372
592
|
/** WCVP taxon id of the matched row (null for unresolved external proposals). */
|
|
373
|
-
taxonId:
|
|
593
|
+
taxonId: z4.string().nullable(),
|
|
374
594
|
/** WCVP taxon id of the resolved accepted taxon (null when unresolved). */
|
|
375
|
-
acceptedTaxonId:
|
|
376
|
-
confidence:
|
|
377
|
-
reason:
|
|
378
|
-
flags:
|
|
379
|
-
targetUrl:
|
|
595
|
+
acceptedTaxonId: z4.string().nullable(),
|
|
596
|
+
confidence: z4.number().nullable(),
|
|
597
|
+
reason: z4.string().nullable(),
|
|
598
|
+
flags: z4.array(z4.string()),
|
|
599
|
+
targetUrl: z4.string().nullable(),
|
|
380
600
|
/** Provider that proposed this candidate (e.g. tnrs, gbif, openrouter,
|
|
381
601
|
* wcvp-local). */
|
|
382
|
-
provider:
|
|
602
|
+
provider: z4.string().nullable()
|
|
383
603
|
});
|
|
384
|
-
var replayAttemptSchema =
|
|
385
|
-
provider:
|
|
386
|
-
status:
|
|
387
|
-
durationMs:
|
|
388
|
-
summary:
|
|
604
|
+
var replayAttemptSchema = z4.object({
|
|
605
|
+
provider: z4.string(),
|
|
606
|
+
status: z4.string(),
|
|
607
|
+
durationMs: z4.number().int().nonnegative(),
|
|
608
|
+
summary: z4.unknown().nullable()
|
|
389
609
|
});
|
|
390
|
-
var replayLayerResultSchema =
|
|
610
|
+
var replayLayerResultSchema = z4.object({
|
|
391
611
|
layer: replayableLayerSchema,
|
|
392
|
-
kind:
|
|
393
|
-
reason:
|
|
612
|
+
kind: z4.enum(["unique", "ambiguous", "miss"]),
|
|
613
|
+
reason: z4.string().nullable(),
|
|
394
614
|
/** Human-facing note when the layer couldn't run as-is (e.g. LLM not
|
|
395
615
|
* configured, no external providers enabled). */
|
|
396
|
-
note:
|
|
397
|
-
providers:
|
|
398
|
-
durationMs:
|
|
616
|
+
note: z4.string().nullable(),
|
|
617
|
+
providers: z4.array(z4.string()),
|
|
618
|
+
durationMs: z4.number().int().nonnegative(),
|
|
399
619
|
/** What was actually submitted to the layer (post gnparser). */
|
|
400
|
-
query:
|
|
401
|
-
canonical:
|
|
402
|
-
authorship:
|
|
403
|
-
inputId:
|
|
620
|
+
query: z4.object({
|
|
621
|
+
canonical: z4.string(),
|
|
622
|
+
authorship: z4.string().nullable(),
|
|
623
|
+
inputId: z4.string().nullable()
|
|
404
624
|
}),
|
|
405
|
-
candidates:
|
|
406
|
-
attempts:
|
|
625
|
+
candidates: z4.array(replayCandidateSchema),
|
|
626
|
+
attempts: z4.array(replayAttemptSchema)
|
|
407
627
|
});
|
|
408
628
|
|
|
409
629
|
// ../shared/src/schemas/auth.ts
|
|
410
|
-
import { z as
|
|
411
|
-
var tokenScopeSchema =
|
|
630
|
+
import { z as z5 } from "zod";
|
|
631
|
+
var tokenScopeSchema = z5.enum([
|
|
412
632
|
"submit:job",
|
|
413
633
|
"read:job",
|
|
414
634
|
"cancel:job",
|
|
@@ -420,77 +640,83 @@ var tokenScopeSchema = z4.enum([
|
|
|
420
640
|
"admin:teams"
|
|
421
641
|
]);
|
|
422
642
|
var ALL_TOKEN_SCOPES = tokenScopeSchema.options;
|
|
423
|
-
var tokenStringSchema =
|
|
424
|
-
var meSchema =
|
|
425
|
-
userId:
|
|
426
|
-
teamId:
|
|
427
|
-
displayName:
|
|
428
|
-
scopes:
|
|
429
|
-
tokenLabel:
|
|
643
|
+
var tokenStringSchema = z5.string().regex(/^ptm_[a-zA-Z0-9]{12}_[a-zA-Z0-9]{32}$/, "Malformed token");
|
|
644
|
+
var meSchema = z5.object({
|
|
645
|
+
userId: z5.string().uuid(),
|
|
646
|
+
teamId: z5.string().uuid(),
|
|
647
|
+
displayName: z5.string(),
|
|
648
|
+
scopes: z5.array(tokenScopeSchema),
|
|
649
|
+
tokenLabel: z5.string().nullable(),
|
|
650
|
+
/** DB id of the token that authenticated this request — matches a row's
|
|
651
|
+
* `id` in the admin token list, so the UI can highlight "this is you". */
|
|
652
|
+
tokenId: z5.string().uuid().nullable()
|
|
430
653
|
});
|
|
431
|
-
var tokenCreateBodySchema =
|
|
432
|
-
label:
|
|
433
|
-
scopes:
|
|
434
|
-
expiresAt:
|
|
654
|
+
var tokenCreateBodySchema = z5.object({
|
|
655
|
+
label: z5.string().min(1).max(120),
|
|
656
|
+
scopes: z5.array(tokenScopeSchema).min(1),
|
|
657
|
+
expiresAt: z5.string().datetime().optional()
|
|
435
658
|
});
|
|
436
|
-
var
|
|
437
|
-
|
|
438
|
-
|
|
439
|
-
|
|
440
|
-
|
|
441
|
-
|
|
442
|
-
|
|
443
|
-
|
|
444
|
-
|
|
659
|
+
var tokenRenameSchema = z5.object({
|
|
660
|
+
label: z5.string().trim().min(1).max(120)
|
|
661
|
+
});
|
|
662
|
+
var tokenSummarySchema = z5.object({
|
|
663
|
+
id: z5.string().uuid(),
|
|
664
|
+
tokenIdPrefix: z5.string(),
|
|
665
|
+
label: z5.string(),
|
|
666
|
+
displayName: z5.string(),
|
|
667
|
+
scopes: z5.array(tokenScopeSchema),
|
|
668
|
+
createdAt: z5.string().datetime(),
|
|
669
|
+
lastUsedAt: z5.string().datetime().nullable(),
|
|
670
|
+
expiresAt: z5.string().datetime().nullable()
|
|
445
671
|
});
|
|
446
672
|
var tokenCreatedSchema = tokenSummarySchema.extend({
|
|
447
673
|
tokenString: tokenStringSchema
|
|
448
674
|
});
|
|
449
|
-
var teamSummarySchema =
|
|
450
|
-
id:
|
|
451
|
-
name:
|
|
452
|
-
createdAt:
|
|
675
|
+
var teamSummarySchema = z5.object({
|
|
676
|
+
id: z5.string().uuid(),
|
|
677
|
+
name: z5.string(),
|
|
678
|
+
createdAt: z5.string().datetime()
|
|
453
679
|
});
|
|
454
|
-
var teamCreateBodySchema =
|
|
455
|
-
name:
|
|
456
|
-
initialUserDisplayName:
|
|
457
|
-
initialToken:
|
|
458
|
-
label:
|
|
459
|
-
scopes:
|
|
680
|
+
var teamCreateBodySchema = z5.object({
|
|
681
|
+
name: z5.string().min(1).max(120),
|
|
682
|
+
initialUserDisplayName: z5.string().min(1).max(120).default("Admin"),
|
|
683
|
+
initialToken: z5.object({
|
|
684
|
+
label: z5.string().min(1).max(120),
|
|
685
|
+
scopes: z5.array(tokenScopeSchema).min(1)
|
|
460
686
|
}).optional()
|
|
461
687
|
});
|
|
462
688
|
var teamCreatedSchema = teamSummarySchema.extend({
|
|
463
|
-
initialUser:
|
|
464
|
-
id:
|
|
465
|
-
displayName:
|
|
689
|
+
initialUser: z5.object({
|
|
690
|
+
id: z5.string().uuid(),
|
|
691
|
+
displayName: z5.string()
|
|
466
692
|
}),
|
|
467
693
|
initialToken: tokenCreatedSchema.nullable()
|
|
468
694
|
});
|
|
469
|
-
var setupStatusSchema =
|
|
470
|
-
needsSetup:
|
|
695
|
+
var setupStatusSchema = z5.object({
|
|
696
|
+
needsSetup: z5.boolean()
|
|
471
697
|
});
|
|
472
|
-
var setupBodySchema =
|
|
473
|
-
teamName:
|
|
474
|
-
displayName:
|
|
698
|
+
var setupBodySchema = z5.object({
|
|
699
|
+
teamName: z5.string().min(1).max(120),
|
|
700
|
+
displayName: z5.string().min(1).max(120).default("Admin")
|
|
475
701
|
});
|
|
476
|
-
var setupResultSchema =
|
|
702
|
+
var setupResultSchema = z5.object({
|
|
477
703
|
team: teamSummarySchema,
|
|
478
|
-
user:
|
|
704
|
+
user: z5.object({ id: z5.string().uuid(), displayName: z5.string() }),
|
|
479
705
|
tokenString: tokenStringSchema,
|
|
480
|
-
scopes:
|
|
706
|
+
scopes: z5.array(tokenScopeSchema)
|
|
481
707
|
});
|
|
482
|
-
var teamLlmSettingsSchema =
|
|
483
|
-
keyConfigured:
|
|
484
|
-
keyHint:
|
|
485
|
-
lightModel:
|
|
486
|
-
heavyModel:
|
|
708
|
+
var teamLlmSettingsSchema = z5.object({
|
|
709
|
+
keyConfigured: z5.boolean(),
|
|
710
|
+
keyHint: z5.string().nullable(),
|
|
711
|
+
lightModel: z5.string(),
|
|
712
|
+
heavyModel: z5.string()
|
|
487
713
|
});
|
|
488
|
-
var teamLlmUpdateSchema =
|
|
489
|
-
openrouterApiKey:
|
|
490
|
-
lightModel:
|
|
491
|
-
heavyModel:
|
|
714
|
+
var teamLlmUpdateSchema = z5.object({
|
|
715
|
+
openrouterApiKey: z5.string().max(400).nullable().optional(),
|
|
716
|
+
lightModel: z5.string().min(1).max(200).optional(),
|
|
717
|
+
heavyModel: z5.string().min(1).max(200).optional()
|
|
492
718
|
});
|
|
493
|
-
var wcvpImportStatusSchema =
|
|
719
|
+
var wcvpImportStatusSchema = z5.enum([
|
|
494
720
|
"queued",
|
|
495
721
|
"downloading",
|
|
496
722
|
"importing",
|
|
@@ -498,68 +724,68 @@ var wcvpImportStatusSchema = z4.enum([
|
|
|
498
724
|
"failed",
|
|
499
725
|
"cancelled"
|
|
500
726
|
]);
|
|
501
|
-
var wcvpImportRunSchema =
|
|
502
|
-
id:
|
|
503
|
-
version:
|
|
504
|
-
sourceUrl:
|
|
727
|
+
var wcvpImportRunSchema = z5.object({
|
|
728
|
+
id: z5.string().uuid(),
|
|
729
|
+
version: z5.string(),
|
|
730
|
+
sourceUrl: z5.string(),
|
|
505
731
|
status: wcvpImportStatusSchema,
|
|
506
|
-
forceOverwrite:
|
|
507
|
-
bytesDownloaded:
|
|
508
|
-
totalBytes:
|
|
509
|
-
insertedCount:
|
|
510
|
-
recordCount:
|
|
511
|
-
message:
|
|
512
|
-
createdAt:
|
|
513
|
-
startedAt:
|
|
514
|
-
completedAt:
|
|
732
|
+
forceOverwrite: z5.boolean(),
|
|
733
|
+
bytesDownloaded: z5.number().int().nonnegative(),
|
|
734
|
+
totalBytes: z5.number().int().nonnegative().nullable(),
|
|
735
|
+
insertedCount: z5.number().int().nonnegative(),
|
|
736
|
+
recordCount: z5.number().int().nonnegative().nullable(),
|
|
737
|
+
message: z5.string().nullable(),
|
|
738
|
+
createdAt: z5.string().datetime(),
|
|
739
|
+
startedAt: z5.string().datetime().nullable(),
|
|
740
|
+
completedAt: z5.string().datetime().nullable()
|
|
515
741
|
});
|
|
516
|
-
var wcvpImportCreateBodySchema =
|
|
517
|
-
version:
|
|
742
|
+
var wcvpImportCreateBodySchema = z5.object({
|
|
743
|
+
version: z5.string().min(1).max(120),
|
|
518
744
|
/** Defaults to the Kew SFTP URL when omitted, so the admin doesn't have
|
|
519
745
|
* to remember it for routine v13/v14 imports. */
|
|
520
|
-
url:
|
|
521
|
-
force:
|
|
746
|
+
url: z5.string().url().optional(),
|
|
747
|
+
force: z5.boolean().optional()
|
|
522
748
|
});
|
|
523
|
-
var adminHealthErrorSchema =
|
|
524
|
-
id:
|
|
525
|
-
action:
|
|
526
|
-
entityType:
|
|
527
|
-
entityId:
|
|
528
|
-
timestamp:
|
|
529
|
-
metadata:
|
|
530
|
-
});
|
|
531
|
-
var adminHealthSchema =
|
|
532
|
-
queueDepth:
|
|
533
|
-
waiting:
|
|
534
|
-
active:
|
|
535
|
-
delayed:
|
|
536
|
-
completed:
|
|
537
|
-
failed:
|
|
538
|
-
paused:
|
|
749
|
+
var adminHealthErrorSchema = z5.object({
|
|
750
|
+
id: z5.string().uuid(),
|
|
751
|
+
action: z5.string(),
|
|
752
|
+
entityType: z5.string(),
|
|
753
|
+
entityId: z5.string().nullable(),
|
|
754
|
+
timestamp: z5.string().datetime(),
|
|
755
|
+
metadata: z5.unknown().nullable()
|
|
756
|
+
});
|
|
757
|
+
var adminHealthSchema = z5.object({
|
|
758
|
+
queueDepth: z5.object({
|
|
759
|
+
waiting: z5.number().int().nonnegative(),
|
|
760
|
+
active: z5.number().int().nonnegative(),
|
|
761
|
+
delayed: z5.number().int().nonnegative(),
|
|
762
|
+
completed: z5.number().int().nonnegative(),
|
|
763
|
+
failed: z5.number().int().nonnegative(),
|
|
764
|
+
paused: z5.number().int().nonnegative()
|
|
539
765
|
}),
|
|
540
|
-
workerCount:
|
|
541
|
-
lastErrors:
|
|
542
|
-
rateLimitBudget:
|
|
543
|
-
max:
|
|
544
|
-
timeWindowSeconds:
|
|
766
|
+
workerCount: z5.number().int().nonnegative(),
|
|
767
|
+
lastErrors: z5.array(adminHealthErrorSchema),
|
|
768
|
+
rateLimitBudget: z5.object({
|
|
769
|
+
max: z5.number().int().nonnegative(),
|
|
770
|
+
timeWindowSeconds: z5.number().int().nonnegative()
|
|
545
771
|
}),
|
|
546
|
-
snapshotVersion:
|
|
547
|
-
snapshotImportedAt:
|
|
772
|
+
snapshotVersion: z5.string().nullable(),
|
|
773
|
+
snapshotImportedAt: z5.string().datetime().nullable(),
|
|
548
774
|
/**
|
|
549
775
|
* Number of cached accepted-name mappings (L0.5 match-history rows) for the
|
|
550
776
|
* caller's team — the size of the team accepted-name cache.
|
|
551
777
|
*/
|
|
552
|
-
teamHistorySize:
|
|
778
|
+
teamHistorySize: z5.number().int().nonnegative(),
|
|
553
779
|
/**
|
|
554
780
|
* Overall (all-teams) tally of which cascade layer produced the winning
|
|
555
781
|
* candidate, across every matched query. `layer` is the raw attempt layer
|
|
556
782
|
* (`L1`…`L7`, `L0.5`, `OVERRIDE`); `count` is the number of matched queries
|
|
557
783
|
* that layer resolved. Descending by count.
|
|
558
784
|
*/
|
|
559
|
-
layerUsage:
|
|
560
|
-
|
|
561
|
-
layer:
|
|
562
|
-
count:
|
|
785
|
+
layerUsage: z5.array(
|
|
786
|
+
z5.object({
|
|
787
|
+
layer: z5.string(),
|
|
788
|
+
count: z5.number().int().nonnegative()
|
|
563
789
|
})
|
|
564
790
|
),
|
|
565
791
|
/**
|
|
@@ -569,18 +795,239 @@ var adminHealthSchema = z4.object({
|
|
|
569
795
|
* Lets the health page surface each external source's usage, hit/miss/error
|
|
570
796
|
* breakdown, latency, and recency — so a misbehaving provider is visible.
|
|
571
797
|
*/
|
|
572
|
-
externalProviders:
|
|
573
|
-
|
|
574
|
-
provider:
|
|
575
|
-
total:
|
|
576
|
-
hits:
|
|
577
|
-
misses:
|
|
578
|
-
errors:
|
|
579
|
-
avgDurationMs:
|
|
580
|
-
lastUsedAt:
|
|
798
|
+
externalProviders: z5.array(
|
|
799
|
+
z5.object({
|
|
800
|
+
provider: z5.string(),
|
|
801
|
+
total: z5.number().int().nonnegative(),
|
|
802
|
+
hits: z5.number().int().nonnegative(),
|
|
803
|
+
misses: z5.number().int().nonnegative(),
|
|
804
|
+
errors: z5.number().int().nonnegative(),
|
|
805
|
+
avgDurationMs: z5.number().int().nonnegative(),
|
|
806
|
+
lastUsedAt: z5.string().datetime().nullable()
|
|
807
|
+
})
|
|
808
|
+
)
|
|
809
|
+
});
|
|
810
|
+
var matchHistoryEntrySchema = z5.object({
|
|
811
|
+
id: z5.string().uuid(),
|
|
812
|
+
/** The normalized input name this mapping is keyed on. */
|
|
813
|
+
normalizedInput: z5.string(),
|
|
814
|
+
acceptedName: z5.string(),
|
|
815
|
+
acceptedIdentifier: z5.object({ namespace: z5.string(), value: z5.string() }).nullable(),
|
|
816
|
+
/** Resolved from the backbone snapshot at read time (null when the stored
|
|
817
|
+
* identifier no longer resolves): authorship + family of the accepted taxon,
|
|
818
|
+
* its identifiers (backbone id, IPNI LSID, portal url) and the portal link. */
|
|
819
|
+
acceptedAuthorship: z5.string().nullable(),
|
|
820
|
+
family: z5.string().nullable(),
|
|
821
|
+
acceptedIdentifiers: z5.array(z5.object({ namespace: z5.string(), value: z5.string() })),
|
|
822
|
+
targetUrl: z5.string().nullable(),
|
|
823
|
+
targetReferential: z5.string(),
|
|
824
|
+
referentialVersion: z5.string(),
|
|
825
|
+
independentUserAcceptanceCount: z5.number().int().nonnegative(),
|
|
826
|
+
rejectionCount: z5.number().int().nonnegative(),
|
|
827
|
+
blacklisted: z5.boolean(),
|
|
828
|
+
needsReconfirmation: z5.boolean(),
|
|
829
|
+
/** Currently usable as a cache hit (not blacklisted, acceptances > rejections). */
|
|
830
|
+
active: z5.boolean(),
|
|
831
|
+
/** Auto-accept (Grade A): acceptances ≥ the team's promotion threshold. */
|
|
832
|
+
promoted: z5.boolean(),
|
|
833
|
+
/** Label of the token that last reviewed this mapping (falls back to the
|
|
834
|
+
* reviewer's display name for rows predating token tracking). */
|
|
835
|
+
lastReviewedBy: z5.string().nullable(),
|
|
836
|
+
firstSeenAt: z5.string().datetime(),
|
|
837
|
+
lastAcceptedAt: z5.string().datetime().nullable()
|
|
838
|
+
});
|
|
839
|
+
var teamCacheResponseSchema = z5.object({
|
|
840
|
+
/** Distinct-user acceptances needed to promote a hit to Grade A. */
|
|
841
|
+
promotionThreshold: z5.number().int().positive(),
|
|
842
|
+
/** Total entries in the team cache (before the limit). */
|
|
843
|
+
total: z5.number().int().nonnegative(),
|
|
844
|
+
entries: z5.array(matchHistoryEntrySchema)
|
|
845
|
+
});
|
|
846
|
+
|
|
847
|
+
// ../shared/src/schemas/backbone.ts
|
|
848
|
+
import { z as z6 } from "zod";
|
|
849
|
+
var backboneSchema = z6.enum(["wcvp", "wfo"]);
|
|
850
|
+
var wfoImportStatusSchema = z6.enum([
|
|
851
|
+
"queued",
|
|
852
|
+
"downloading",
|
|
853
|
+
"importing",
|
|
854
|
+
"completed",
|
|
855
|
+
"failed",
|
|
856
|
+
"cancelled"
|
|
857
|
+
]);
|
|
858
|
+
var wfoImportRunSchema = z6.object({
|
|
859
|
+
id: z6.string().uuid(),
|
|
860
|
+
version: z6.string(),
|
|
861
|
+
sourceUrl: z6.string(),
|
|
862
|
+
status: wfoImportStatusSchema,
|
|
863
|
+
forceOverwrite: z6.boolean(),
|
|
864
|
+
bytesDownloaded: z6.number().int().nonnegative(),
|
|
865
|
+
totalBytes: z6.number().int().nonnegative().nullable(),
|
|
866
|
+
insertedCount: z6.number().int().nonnegative(),
|
|
867
|
+
recordCount: z6.number().int().nonnegative().nullable(),
|
|
868
|
+
message: z6.string().nullable(),
|
|
869
|
+
createdAt: z6.string().datetime(),
|
|
870
|
+
startedAt: z6.string().datetime().nullable(),
|
|
871
|
+
completedAt: z6.string().datetime().nullable()
|
|
872
|
+
});
|
|
873
|
+
var wfoImportCreateBodySchema = z6.object({
|
|
874
|
+
version: z6.string().min(1).max(120),
|
|
875
|
+
/** Direct download URL (from the Zenodo version listing). Required for WFO
|
|
876
|
+
* since versions aren't derivable from the label like WCVP's Kew URLs. */
|
|
877
|
+
url: z6.string().url().optional(),
|
|
878
|
+
force: z6.boolean().optional()
|
|
879
|
+
});
|
|
880
|
+
var snapshotKindSchema = z6.enum(["official", "derived"]);
|
|
881
|
+
var wfoSnapshotSummarySchema = z6.object({
|
|
882
|
+
version: z6.string(),
|
|
883
|
+
recordCount: z6.number().int().nonnegative(),
|
|
884
|
+
importedAt: z6.string().datetime(),
|
|
885
|
+
isLatest: z6.boolean(),
|
|
886
|
+
kind: snapshotKindSchema,
|
|
887
|
+
baseVersion: z6.string().nullable(),
|
|
888
|
+
label: z6.string().nullable(),
|
|
889
|
+
ownerTeamId: z6.string().nullable()
|
|
890
|
+
});
|
|
891
|
+
var wfoVersionSchema = z6.object({
|
|
892
|
+
/** Release label, e.g. "2025-12". */
|
|
893
|
+
version: z6.string(),
|
|
894
|
+
/** Zenodo record id for this version. */
|
|
895
|
+
recordId: z6.string(),
|
|
896
|
+
/** The plant-list archive filename inside the record. */
|
|
897
|
+
fileName: z6.string(),
|
|
898
|
+
/** Direct content download URL. */
|
|
899
|
+
url: z6.string().url(),
|
|
900
|
+
sizeBytes: z6.number().int().nonnegative().nullable(),
|
|
901
|
+
publishedAt: z6.string().nullable(),
|
|
902
|
+
isLatest: z6.boolean()
|
|
903
|
+
});
|
|
904
|
+
var wfoVersionListSchema = z6.array(wfoVersionSchema);
|
|
905
|
+
var backboneKeySchema = z6.enum(["wcvp", "wfo"]);
|
|
906
|
+
var backboneDefaultsSchema = z6.object({
|
|
907
|
+
wcvp: z6.string().nullable(),
|
|
908
|
+
wfo: z6.string().nullable()
|
|
909
|
+
});
|
|
910
|
+
var backboneDefaultUpdateSchema = z6.object({
|
|
911
|
+
backbone: backboneKeySchema,
|
|
912
|
+
version: z6.string().min(1).nullable()
|
|
913
|
+
});
|
|
914
|
+
var DERIVED_SLUG_RE = /^[a-z0-9][a-z0-9-]{1,40}$/;
|
|
915
|
+
var MAX_BASE_VERSION_LENGTH = 120;
|
|
916
|
+
var MAX_SNAPSHOT_VERSION_LENGTH = MAX_BASE_VERSION_LENGTH + 1 + 41;
|
|
917
|
+
var derivedUploadMetaSchema = z6.object({
|
|
918
|
+
baseVersion: z6.string().min(1).max(MAX_BASE_VERSION_LENGTH),
|
|
919
|
+
slug: z6.string().regex(DERIVED_SLUG_RE, "slug: lowercase letters, digits and dashes (2\u201341 chars)"),
|
|
920
|
+
label: z6.string().trim().min(1).max(120),
|
|
921
|
+
notes: z6.string().trim().max(2e3).optional()
|
|
922
|
+
});
|
|
923
|
+
var derivedUploadStatusSchema = z6.enum([
|
|
924
|
+
"staging",
|
|
925
|
+
"ready",
|
|
926
|
+
"invalid",
|
|
927
|
+
"importing",
|
|
928
|
+
"imported",
|
|
929
|
+
"failed",
|
|
930
|
+
"discarded"
|
|
931
|
+
]);
|
|
932
|
+
var derivedIssueSeveritySchema = z6.enum(["error", "warning"]);
|
|
933
|
+
var derivedIssueSchema = z6.object({
|
|
934
|
+
code: z6.string(),
|
|
935
|
+
severity: derivedIssueSeveritySchema,
|
|
936
|
+
count: z6.number().int().nonnegative(),
|
|
937
|
+
/** Free-form detail for the UI (e.g. the offending vocabulary values). */
|
|
938
|
+
detail: z6.string().nullable(),
|
|
939
|
+
samples: z6.array(
|
|
940
|
+
z6.object({
|
|
941
|
+
taxonId: z6.string().nullable(),
|
|
942
|
+
name: z6.string().nullable(),
|
|
943
|
+
detail: z6.string().nullable(),
|
|
944
|
+
line: z6.number().int().nullable()
|
|
581
945
|
})
|
|
582
946
|
)
|
|
583
947
|
});
|
|
948
|
+
var derivedDiffFieldSchema = z6.enum([
|
|
949
|
+
"canonical_name",
|
|
950
|
+
"scientific_name",
|
|
951
|
+
"authorship",
|
|
952
|
+
"rank",
|
|
953
|
+
"taxonomic_status",
|
|
954
|
+
"accepted_name_usage_id",
|
|
955
|
+
"parent_name_usage_id",
|
|
956
|
+
"family"
|
|
957
|
+
]);
|
|
958
|
+
var DERIVED_DIFF_FIELDS = derivedDiffFieldSchema.options;
|
|
959
|
+
var diffSampleSchema = z6.object({ taxonId: z6.string(), name: z6.string().nullable() });
|
|
960
|
+
var changedSampleSchema = diffSampleSchema.extend({
|
|
961
|
+
changes: z6.array(
|
|
962
|
+
z6.object({ field: derivedDiffFieldSchema, before: z6.string().nullable(), after: z6.string().nullable() })
|
|
963
|
+
)
|
|
964
|
+
});
|
|
965
|
+
var derivedReportSchema = z6.object({
|
|
966
|
+
rowsRead: z6.number().int().nonnegative(),
|
|
967
|
+
rowsStaged: z6.number().int().nonnegative(),
|
|
968
|
+
rowsRejected: z6.number().int().nonnegative(),
|
|
969
|
+
baseRowCount: z6.number().int().nonnegative(),
|
|
970
|
+
errors: z6.number().int().nonnegative(),
|
|
971
|
+
warnings: z6.number().int().nonnegative(),
|
|
972
|
+
issues: z6.array(derivedIssueSchema),
|
|
973
|
+
diff: z6.object({
|
|
974
|
+
added: z6.object({ count: z6.number().int().nonnegative(), samples: z6.array(diffSampleSchema) }),
|
|
975
|
+
removed: z6.object({ count: z6.number().int().nonnegative(), samples: z6.array(diffSampleSchema) }),
|
|
976
|
+
changed: z6.object({
|
|
977
|
+
count: z6.number().int().nonnegative(),
|
|
978
|
+
byField: z6.record(z6.string(), z6.number().int().nonnegative()),
|
|
979
|
+
samples: z6.array(changedSampleSchema)
|
|
980
|
+
}),
|
|
981
|
+
unchanged: z6.number().int().nonnegative()
|
|
982
|
+
})
|
|
983
|
+
});
|
|
984
|
+
var derivedUploadSchema = z6.object({
|
|
985
|
+
id: z6.string().uuid(),
|
|
986
|
+
backbone: backboneSchema,
|
|
987
|
+
baseVersion: z6.string(),
|
|
988
|
+
slug: z6.string(),
|
|
989
|
+
label: z6.string(),
|
|
990
|
+
notes: z6.string().nullable(),
|
|
991
|
+
/** The snapshot version this upload installs as (`<base>+<slug>`). */
|
|
992
|
+
version: z6.string(),
|
|
993
|
+
originalFilename: z6.string().nullable(),
|
|
994
|
+
sha256: z6.string(),
|
|
995
|
+
sizeBytes: z6.number().int().nonnegative(),
|
|
996
|
+
status: derivedUploadStatusSchema,
|
|
997
|
+
rowsRead: z6.number().int().nonnegative(),
|
|
998
|
+
report: derivedReportSchema.nullable(),
|
|
999
|
+
message: z6.string().nullable(),
|
|
1000
|
+
createdAt: z6.string().datetime(),
|
|
1001
|
+
expiresAt: z6.string().datetime(),
|
|
1002
|
+
importedVersion: z6.string().nullable()
|
|
1003
|
+
});
|
|
1004
|
+
var derivedConfirmBodySchema = z6.object({
|
|
1005
|
+
/** Acknowledge warnings. Never overrides errors. */
|
|
1006
|
+
force: z6.boolean().optional()
|
|
1007
|
+
});
|
|
1008
|
+
|
|
1009
|
+
// ../shared/src/schemas/openrouter.ts
|
|
1010
|
+
import { z as z7 } from "zod";
|
|
1011
|
+
var openRouterModelSchema = z7.object({
|
|
1012
|
+
/** Full model id, e.g. `anthropic/claude-opus-4.8` or the alias
|
|
1013
|
+
* `~anthropic/claude-haiku-latest`. Used verbatim as the OpenRouter model. */
|
|
1014
|
+
id: z7.string(),
|
|
1015
|
+
/** Human label from OpenRouter (falls back to the id). */
|
|
1016
|
+
name: z7.string(),
|
|
1017
|
+
/** Author slug (the part before `/`), with any leading `~` stripped. */
|
|
1018
|
+
author: z7.string(),
|
|
1019
|
+
/** Unix seconds the model was published; used to rank "latest". */
|
|
1020
|
+
created: z7.number(),
|
|
1021
|
+
/** Context window in tokens, when OpenRouter reports it. */
|
|
1022
|
+
contextLength: z7.number().nullable(),
|
|
1023
|
+
/** USD price per PROMPT token (input). Null when OpenRouter doesn't report a
|
|
1024
|
+
* numeric price. 0 = free. */
|
|
1025
|
+
promptPriceUsd: z7.number().nullable(),
|
|
1026
|
+
/** USD price per COMPLETION token (output). */
|
|
1027
|
+
completionPriceUsd: z7.number().nullable(),
|
|
1028
|
+
/** An auto-updating `…-latest` pointer (e.g. `~anthropic/claude-haiku-latest`). */
|
|
1029
|
+
isAlias: z7.boolean()
|
|
1030
|
+
});
|
|
584
1031
|
|
|
585
1032
|
// ../shared/src/normalize.ts
|
|
586
1033
|
var NULL_SENTINELS = /* @__PURE__ */ new Set(["", "na", "n/a", "null", "-", "\u2014", "unknown", "undet", "undet."]);
|
|
@@ -708,21 +1155,21 @@ function detectIdTypeDistribution(samples, opts) {
|
|
|
708
1155
|
}
|
|
709
1156
|
|
|
710
1157
|
// ../shared/src/column-map.ts
|
|
711
|
-
import { z as
|
|
712
|
-
var columnMappingSchema =
|
|
713
|
-
nameColumn:
|
|
714
|
-
idColumn:
|
|
715
|
-
familyColumn:
|
|
716
|
-
genusColumn:
|
|
717
|
-
rankColumn:
|
|
718
|
-
authorColumn:
|
|
719
|
-
});
|
|
720
|
-
var detectColumnsBodySchema =
|
|
721
|
-
headers:
|
|
722
|
-
});
|
|
723
|
-
var detectColumnsResponseSchema =
|
|
1158
|
+
import { z as z8 } from "zod";
|
|
1159
|
+
var columnMappingSchema = z8.object({
|
|
1160
|
+
nameColumn: z8.string().nullable(),
|
|
1161
|
+
idColumn: z8.string().nullable(),
|
|
1162
|
+
familyColumn: z8.string().nullable(),
|
|
1163
|
+
genusColumn: z8.string().nullable(),
|
|
1164
|
+
rankColumn: z8.string().nullable(),
|
|
1165
|
+
authorColumn: z8.string().nullable()
|
|
1166
|
+
});
|
|
1167
|
+
var detectColumnsBodySchema = z8.object({
|
|
1168
|
+
headers: z8.array(z8.string().min(1)).min(1).max(200)
|
|
1169
|
+
});
|
|
1170
|
+
var detectColumnsResponseSchema = z8.object({
|
|
724
1171
|
mapping: columnMappingSchema,
|
|
725
|
-
usedLlm:
|
|
1172
|
+
usedLlm: z8.boolean()
|
|
726
1173
|
});
|
|
727
1174
|
|
|
728
1175
|
// src/api-client.ts
|
|
@@ -775,7 +1222,7 @@ var apiClient = {
|
|
|
775
1222
|
const buf = await fs.readFile(filePath);
|
|
776
1223
|
const fd = new FormData();
|
|
777
1224
|
const lower = filePath.toLowerCase();
|
|
778
|
-
const mime = lower.endsWith(".json") ? "application/json" : "text/csv";
|
|
1225
|
+
const mime = lower.endsWith(".json") ? "application/json" : lower.endsWith(".xlsx") || lower.endsWith(".xls") ? "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" : "text/csv";
|
|
779
1226
|
fd.set("file", new Blob([buf], { type: mime }), basename(filePath));
|
|
780
1227
|
fd.set("config", JSON.stringify(config));
|
|
781
1228
|
if (name && name.trim()) fd.set("name", name.trim());
|
|
@@ -793,6 +1240,7 @@ var apiClient = {
|
|
|
793
1240
|
async downloadJob(creds, id, opts) {
|
|
794
1241
|
const params = new URLSearchParams({ format: opts.format });
|
|
795
1242
|
if (opts.confirmedOnly) params.set("confirmedOnly", "true");
|
|
1243
|
+
if (opts.dedupe) params.set("dedupe", "true");
|
|
796
1244
|
if (opts.bundle) params.set("bundle", "true");
|
|
797
1245
|
if (opts.delimiter) params.set("delimiter", opts.delimiter);
|
|
798
1246
|
if (opts.columns) params.set("columns", opts.columns);
|
|
@@ -1023,8 +1471,9 @@ function trunc(s, n) {
|
|
|
1023
1471
|
}
|
|
1024
1472
|
|
|
1025
1473
|
// src/index.ts
|
|
1474
|
+
var { version } = createRequire(import.meta.url)("../package.json");
|
|
1026
1475
|
var program = new Command();
|
|
1027
|
-
program.name("planttaxomatcher").description("PlantTaxoMatcher CLI").version(
|
|
1476
|
+
program.name("planttaxomatcher").description("PlantTaxoMatcher CLI").version(version);
|
|
1028
1477
|
program.command("login").description("Save a personal token + server URL").option(
|
|
1029
1478
|
"--token <token>",
|
|
1030
1479
|
"Personal token (ptm_...). Avoid on shared hosts: it is visible in shell history and the process list. Prefer --token-stdin or the PLANTTAXOMATCHER_TOKEN env var."
|
|
@@ -1036,17 +1485,15 @@ program.command("login").description("Save a personal token + server URL").optio
|
|
|
1036
1485
|
"--insecure",
|
|
1037
1486
|
"Allow sending the token over cleartext http to a non-loopback server (NOT recommended).",
|
|
1038
1487
|
false
|
|
1039
|
-
).action(
|
|
1040
|
-
|
|
1041
|
-
|
|
1042
|
-
|
|
1043
|
-
|
|
1044
|
-
throw new Error("malformed token \u2014 expected ptm_<12 chars>_<32 chars>");
|
|
1045
|
-
}
|
|
1046
|
-
await writeCredentials({ token, server: opts.server });
|
|
1047
|
-
console.log(kleur2.green("\u2713"), `Logged in to ${opts.server}`);
|
|
1488
|
+
).action(async (opts) => {
|
|
1489
|
+
assertServerTransport(opts.server, !!opts.insecure);
|
|
1490
|
+
const token = await resolveLoginToken(opts);
|
|
1491
|
+
if (!tokenStringSchema.safeParse(token).success) {
|
|
1492
|
+
throw new Error("malformed token \u2014 expected ptm_<12 chars>_<32 chars>");
|
|
1048
1493
|
}
|
|
1049
|
-
);
|
|
1494
|
+
await writeCredentials({ token, server: opts.server });
|
|
1495
|
+
console.log(kleur2.green("\u2713"), `Logged in to ${opts.server}`);
|
|
1496
|
+
});
|
|
1050
1497
|
program.command("logout").description("Clear saved credentials").action(async () => {
|
|
1051
1498
|
await clearCredentials();
|
|
1052
1499
|
console.log(kleur2.green("\u2713"), "Logged out");
|
|
@@ -1055,6 +1502,7 @@ program.command("whoami").description("Show current user").action(async () => {
|
|
|
1055
1502
|
const creds = await requireCredentials();
|
|
1056
1503
|
const me = await apiClient.me(creds);
|
|
1057
1504
|
console.log(`${me.displayName} team=${me.teamId} scopes=${me.scopes.join(",")}`);
|
|
1505
|
+
console.log(`server=${creds.server}`);
|
|
1058
1506
|
});
|
|
1059
1507
|
program.command("list").description("List recent jobs").action(async () => {
|
|
1060
1508
|
const creds = await requireCredentials();
|
|
@@ -1080,25 +1528,27 @@ program.command("submit <files...>").description(
|
|
|
1080
1528
|
"--name <label>",
|
|
1081
1529
|
"Job name override (single file only; ignored when multiple files match \u2014 the file name is used)"
|
|
1082
1530
|
).requiredOption("--name-column <name>", "Column holding the scientific name").option("--id-column <name>", "Column holding a known ID").option("--family-column <name>", "Column holding family").option("--genus-column <name>", "Column holding genus").option("--rank-column <name>", "Column holding rank").option("--author-column <name>", "Column holding authorship").option("--filter-column <name>", "Only process rows where this column matches --filter-value; others are skipped").option("--filter-value <value>", "Value the --filter-column must equal (trimmed, case-insensitive)").option("--author-mode <mode>", "ignore|prefer|strict", "prefer").option("--parallel <n>", "Per-job parallelism", "4").option("--review-mode <mode>", "off|recommended|strict", "recommended").option(
|
|
1083
|
-
"--
|
|
1084
|
-
"
|
|
1085
|
-
false
|
|
1086
|
-
).option("--allow-llm", "Allow Layer 7 LLM cascade (breaks ties between competing matches)", false).option("--llm-cap-cents <cents>", "Max LLM spend in cents", "500").option("--referential <version>", "WCVP snapshot version to match against (default: latest)").option("--no-watch", "Do not stream progress after submit").option(
|
|
1087
|
-
"--dry-run",
|
|
1088
|
-
"Preview locally (normalize first rows + detect ID type) and confirm before uploading",
|
|
1531
|
+
"--species-level",
|
|
1532
|
+
"Roll infraspecific accepted taxa (varieties, subspecies, forms) up to their species, so every output row sits at species level",
|
|
1089
1533
|
false
|
|
1090
1534
|
).option(
|
|
1091
|
-
"--
|
|
1092
|
-
"
|
|
1093
|
-
|
|
1094
|
-
).
|
|
1535
|
+
"--ignore-author",
|
|
1536
|
+
"Accept a unique canonical name match whose author couldn't be confirmed instead of sending it to review (an author that CONFLICTS with the matched taxon still reviews)",
|
|
1537
|
+
false
|
|
1538
|
+
).option("--keep-infraspecific", "Deprecated \u2014 now the default; has no effect", false).option("--allow-llm", "Allow Layer 7 LLM cascade (breaks ties between competing matches)", false).option("--llm-cap-cents <cents>", "Max LLM spend in cents", "500").option(
|
|
1539
|
+
"--referential <version>",
|
|
1540
|
+
"WCVP snapshot version to match against (default: your team's configured default snapshot, else the newest import)"
|
|
1541
|
+
).option("--no-watch", "Do not stream progress after submit").option("--dry-run", "Preview locally (normalize first rows + detect ID type) and confirm before uploading", false).option("--dry-run-rows <n>", "Number of rows to show in the dry-run preview table (default 10)", "10").action(async (files, opts) => {
|
|
1095
1542
|
const creds = await requireCredentials();
|
|
1096
1543
|
const inputs = await expandInputs(files);
|
|
1097
1544
|
if (inputs.length === 0) throw new Error("no input files");
|
|
1098
1545
|
if (opts.name && inputs.length > 1) {
|
|
1546
|
+
console.log(kleur2.yellow("!"), "--name ignored for multi-file submit; using each file name as the job name");
|
|
1547
|
+
}
|
|
1548
|
+
if (opts.keepInfraspecific) {
|
|
1099
1549
|
console.log(
|
|
1100
1550
|
kleur2.yellow("!"),
|
|
1101
|
-
"--
|
|
1551
|
+
"--keep-infraspecific is deprecated and does nothing \u2014 infraspecific taxa are kept by default. Pass --species-level to roll them up to the species."
|
|
1102
1552
|
);
|
|
1103
1553
|
}
|
|
1104
1554
|
if (inputs.length > 1) {
|
|
@@ -1144,9 +1594,12 @@ ${file}`));
|
|
|
1144
1594
|
allowLlm: !!opts.allowLlm,
|
|
1145
1595
|
llmCostCapCents: Number(opts.llmCapCents ?? 500),
|
|
1146
1596
|
reviewMode: String(opts.reviewMode ?? "recommended"),
|
|
1147
|
-
//
|
|
1148
|
-
//
|
|
1149
|
-
speciesLevelAcceptedOnly:
|
|
1597
|
+
// Opt-in: by default an infraspecific accepted taxon is kept at the
|
|
1598
|
+
// rank it resolved to, not rolled up to its species.
|
|
1599
|
+
speciesLevelAcceptedOnly: !!opts.speciesLevel,
|
|
1600
|
+
// Opt-in: an unconfirmed author still routes the match to review
|
|
1601
|
+
// unless the caller declares the author unimportant.
|
|
1602
|
+
acceptUnconfirmedAuthor: !!opts.ignoreAuthor,
|
|
1150
1603
|
exportConfirmedOnly: false,
|
|
1151
1604
|
forceReviewFamilies: [],
|
|
1152
1605
|
...opts.referential ? { referentialVersion: String(opts.referential) } : {}
|
|
@@ -1186,10 +1639,7 @@ program.command("cancel <jobId>").description("Cancel a job. Already-matched row
|
|
|
1186
1639
|
const job = await apiClient.cancelJob(creds, jobId);
|
|
1187
1640
|
console.log(kleur2.green("\u2713"), `cancelled: ${job.id} (status=${job.status})`);
|
|
1188
1641
|
});
|
|
1189
|
-
program.command("download <jobId>").description("Download a job export (CSV, JSON, or NDJSON; optionally bundled with NOTICE.md)").option("--format <fmt>", "csv | xlsx | json | ndjson", "csv").option("--confirmed-only", "only matched rows that are accepted / not pending", false).option("--delimiter <sep>", "CSV separator: comma | semicolon | tab | pipe (csv format only)", "comma").option("--bundle", "wrap in a ZIP with NOTICE.md citing the WCVP snapshot + providers", false).option(
|
|
1190
|
-
"--columns <list>",
|
|
1191
|
-
"comma-separated result/upload column keys to KEEP (default: all). See --list-columns"
|
|
1192
|
-
).option(
|
|
1642
|
+
program.command("download <jobId>").description("Download a job export (CSV, JSON, or NDJSON; optionally bundled with NOTICE.md)").option("--format <fmt>", "csv | xlsx | json | ndjson", "csv").option("--confirmed-only", "only matched rows that are accepted / not pending", false).option("--dedupe", "collapse rows that resolved to the same accepted taxon to one line", false).option("--delimiter <sep>", "CSV separator: comma | semicolon | tab | pipe (csv format only)", "comma").option("--bundle", "wrap in a ZIP with NOTICE.md citing the WCVP snapshot + providers", false).option("--columns <list>", "comma-separated result/upload column keys to KEEP (default: all). See --list-columns").option(
|
|
1193
1643
|
"--wcvp-extra <list>",
|
|
1194
1644
|
"comma-separated extra WCVP fields to append as wcvp_<key> columns (e.g. ipni_id,powo_id). See --list-columns"
|
|
1195
1645
|
).option("--list-columns", "print the columns available for this job and exit", false).option("--output <path>", "write to this path; default is the server-provided filename in CWD").action(async (jobId, opts) => {
|
|
@@ -1211,6 +1661,7 @@ program.command("download <jobId>").description("Download a job export (CSV, JSO
|
|
|
1211
1661
|
throw new Error(`--format must be csv, xlsx, json or ndjson (got ${String(format)})`);
|
|
1212
1662
|
}
|
|
1213
1663
|
const confirmedOnly = !!opts.confirmedOnly;
|
|
1664
|
+
const dedupe = !!opts.dedupe;
|
|
1214
1665
|
const bundle = !!opts.bundle;
|
|
1215
1666
|
const delimiter = String(opts.delimiter ?? "comma");
|
|
1216
1667
|
if (!["comma", "semicolon", "tab", "pipe"].includes(delimiter)) {
|
|
@@ -1219,6 +1670,7 @@ program.command("download <jobId>").description("Download a job export (CSV, JSO
|
|
|
1219
1670
|
const { filename, body } = await apiClient.downloadJob(creds, jobId, {
|
|
1220
1671
|
format,
|
|
1221
1672
|
confirmedOnly,
|
|
1673
|
+
dedupe,
|
|
1222
1674
|
bundle,
|
|
1223
1675
|
...format === "csv" && delimiter !== "comma" ? { delimiter } : {},
|
|
1224
1676
|
...opts.columns ? { columns: String(opts.columns) } : {},
|
|
@@ -1268,9 +1720,7 @@ async function resolveLoginToken(opts) {
|
|
|
1268
1720
|
const env = process.env.PLANTTAXOMATCHER_TOKEN;
|
|
1269
1721
|
if (env && env.trim()) return env.trim();
|
|
1270
1722
|
if (!process.stdin.isTTY) {
|
|
1271
|
-
throw new Error(
|
|
1272
|
-
"no token provided. Pass --token-stdin, set PLANTTAXOMATCHER_TOKEN, or run interactively."
|
|
1273
|
-
);
|
|
1723
|
+
throw new Error("no token provided. Pass --token-stdin, set PLANTTAXOMATCHER_TOKEN, or run interactively.");
|
|
1274
1724
|
}
|
|
1275
1725
|
const entered = await password({ message: "Personal token (ptm_...)", mask: true });
|
|
1276
1726
|
return entered.trim();
|
|
@@ -1297,6 +1747,7 @@ async function expandInputs(patterns) {
|
|
|
1297
1747
|
}
|
|
1298
1748
|
async function streamJob(creds, jobId) {
|
|
1299
1749
|
console.log(kleur2.cyan("\u2192"), `Streaming progress for ${jobId} \u2026`);
|
|
1750
|
+
let lastLineWidth = 0;
|
|
1300
1751
|
for await (const evt of apiClient.streamJob(creds, jobId)) {
|
|
1301
1752
|
const t = String(evt.type ?? "");
|
|
1302
1753
|
if (t === "heartbeat") continue;
|
|
@@ -1314,7 +1765,10 @@ async function streamJob(creds, jobId) {
|
|
|
1314
1765
|
} else if (t === "progress") {
|
|
1315
1766
|
const p = evt.processedQueries ?? evt.processedRows ?? 0;
|
|
1316
1767
|
const total = evt.totalQueries ?? evt.totalRows ?? 0;
|
|
1317
|
-
|
|
1768
|
+
const phase = p === 0 && typeof evt.phase === "string" ? ` (${evt.phase}\u2026)` : "";
|
|
1769
|
+
const line = ` progress: ${p}/${total}${phase}`;
|
|
1770
|
+
process.stdout.write(`\r${line.padEnd(lastLineWidth)}`);
|
|
1771
|
+
lastLineWidth = line.length;
|
|
1318
1772
|
} else if (t === "completed") {
|
|
1319
1773
|
process.stdout.write("\n");
|
|
1320
1774
|
console.log(kleur2.green("\u2713"), "completed");
|