@plantnet/planttaxomatcher 0.1.3 → 0.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/dist/index.js +863 -410
  2. package/package.json +42 -43
package/dist/index.js CHANGED
@@ -1,6 +1,7 @@
1
1
  #!/usr/bin/env node
2
2
 
3
3
  // src/index.ts
4
+ import { createRequire } from "module";
4
5
  import { promises as fs4 } from "fs";
5
6
  import { parse as parsePath } from "path";
6
7
  import { confirm, password } from "@inquirer/prompts";
@@ -31,7 +32,9 @@ var evidenceTypeSchema = z2.enum([
31
32
  "local_fuzzy",
32
33
  "external_fuzzy",
33
34
  "team_history",
34
- "llm"
35
+ "llm",
36
+ // Resolved via the OTHER backbone (WFO↔WCVP) then mapped back to the target.
37
+ "cross_backbone"
35
38
  ]);
36
39
  var gradeSchema = z2.enum(["A", "B", "C"]);
37
40
  var reviewStatusSchema = z2.enum(["not_required", "pending", "accepted", "rejected", "overridden"]);
@@ -59,8 +62,80 @@ var scoreBreakdownSchema = z2.object({
59
62
  });
60
63
 
61
64
  // ../shared/src/schemas/job.ts
65
+ import { z as z4 } from "zod";
66
+
67
+ // ../shared/src/schemas/plugins.ts
62
68
  import { z as z3 } from "zod";
63
- var jobStatusSchema = z3.enum([
69
+ var verifierStateSchema = z3.enum(["pass", "fail", "error", "skipped"]);
70
+ var verifierColumnKindSchema = z3.enum(["check", "badge", "link", "text"]);
71
+ var verifierColumnSchema = z3.object({
72
+ /** Export column id, e.g. `verify_plantnet`. */
73
+ key: z3.string(),
74
+ /** Human-facing header. */
75
+ header: z3.string(),
76
+ kind: verifierColumnKindSchema
77
+ });
78
+ var pluginRunStatusSchema = z3.enum(["queued", "running", "completed", "failed", "cancelled"]);
79
+ var pluginRunTriggerSchema = z3.enum(["auto", "manual"]);
80
+ var pluginConfigOptionSchema = z3.object({
81
+ value: z3.string(),
82
+ label: z3.string()
83
+ });
84
+ var pluginConfigFieldSchema = z3.object({
85
+ key: z3.string(),
86
+ label: z3.string(),
87
+ type: z3.literal("select"),
88
+ options: z3.array(pluginConfigOptionSchema).min(1),
89
+ default: z3.string()
90
+ });
91
+ var pluginDescriptorSchema = z3.object({
92
+ id: z3.string(),
93
+ version: z3.string(),
94
+ title: z3.string(),
95
+ description: z3.string(),
96
+ /** Deployment can actually run it (key/env present). */
97
+ available: z3.boolean(),
98
+ columns: z3.array(verifierColumnSchema),
99
+ configFields: z3.array(pluginConfigFieldSchema)
100
+ });
101
+ var pluginRunRequestSchema = z3.object({
102
+ config: z3.record(z3.string(), z3.string()).default({})
103
+ });
104
+ var pluginRunSummarySchema = z3.object({
105
+ id: z3.string().uuid(),
106
+ jobId: z3.string().ulid(),
107
+ pluginId: z3.string(),
108
+ pluginVersion: z3.string(),
109
+ config: z3.record(z3.string(), z3.unknown()),
110
+ status: pluginRunStatusSchema,
111
+ trigger: pluginRunTriggerSchema,
112
+ totalUnits: z3.number().int().nonnegative(),
113
+ processedUnits: z3.number().int().nonnegative(),
114
+ passUnits: z3.number().int().nonnegative(),
115
+ errorUnits: z3.number().int().nonnegative(),
116
+ /** Label of the token that triggered a manual run; null for auto runs. */
117
+ triggeredBy: z3.string().nullable(),
118
+ /** `failed` run: why it stopped. `completed` run: why its errored rows
119
+ * couldn't be checked (most frequent reasons first). */
120
+ error: z3.string().nullable(),
121
+ createdAt: z3.string().datetime(),
122
+ startedAt: z3.string().datetime().nullable(),
123
+ completedAt: z3.string().datetime().nullable()
124
+ });
125
+ var pluginRunsResponseSchema = z3.object({
126
+ runs: z3.array(pluginRunSummarySchema)
127
+ });
128
+ var jobRowPluginResultSchema = z3.object({
129
+ pluginId: z3.string(),
130
+ state: verifierStateSchema,
131
+ label: z3.string().nullable(),
132
+ url: z3.string().nullable(),
133
+ /** Why this verdict — the error reason, or what a `fail` was checked against. */
134
+ detail: z3.string().nullable()
135
+ });
136
+
137
+ // ../shared/src/schemas/job.ts
138
+ var jobStatusSchema = z4.enum([
64
139
  "queued",
65
140
  "parsing",
66
141
  "matching",
@@ -69,39 +144,87 @@ var jobStatusSchema = z3.enum([
69
144
  "completed",
70
145
  "failed"
71
146
  ]);
72
- var authorModeSchema = z3.enum(["ignore", "prefer", "strict"]);
73
- var reviewModeSchema = z3.enum(["off", "recommended", "strict"]);
74
- var jobConfigSchema = z3.object({
75
- nameColumn: z3.string().min(1),
76
- idColumn: z3.string().nullable().optional(),
77
- familyColumn: z3.string().nullable().optional(),
78
- genusColumn: z3.string().nullable().optional(),
79
- rankColumn: z3.string().nullable().optional(),
80
- authorColumn: z3.string().nullable().optional(),
81
- sourceReferentialColumn: z3.string().nullable().optional(),
82
- sourceIdColumn: z3.string().nullable().optional(),
147
+ var authorModeSchema = z4.enum(["ignore", "prefer", "strict"]);
148
+ var reviewModeSchema = z4.enum(["off", "recommended", "strict"]);
149
+ var jobConfigSchema = z4.object({
150
+ nameColumn: z4.string().min(1),
151
+ idColumn: z4.string().nullable().optional(),
152
+ familyColumn: z4.string().nullable().optional(),
153
+ genusColumn: z4.string().nullable().optional(),
154
+ rankColumn: z4.string().nullable().optional(),
155
+ authorColumn: z4.string().nullable().optional(),
156
+ sourceReferentialColumn: z4.string().nullable().optional(),
157
+ sourceIdColumn: z4.string().nullable().optional(),
83
158
  authorMode: authorModeSchema.default("prefer"),
84
- matchAuthors: z3.boolean().default(true),
85
- parallelism: z3.number().int().min(1).max(10).default(4),
86
- allowFuzzy: z3.boolean().default(true),
87
- allowLlm: z3.boolean().default(false),
88
- llmCostCapCents: z3.number().int().min(0).default(500),
159
+ matchAuthors: z4.boolean().default(true),
160
+ parallelism: z4.number().int().min(1).max(10).default(4),
161
+ allowFuzzy: z4.boolean().default(true),
162
+ allowLlm: z4.boolean().default(false),
163
+ llmCostCapCents: z4.number().int().min(0).default(500),
89
164
  reviewMode: reviewModeSchema.default("recommended"),
90
- exportConfirmedOnly: z3.boolean().default(false),
165
+ exportConfirmedOnly: z4.boolean().default(false),
166
+ /**
167
+ * When false, this job's review decisions (accept + reject) do NOT feed the
168
+ * team's Layer 0.5 match-history cache — a one-off or experimental job can't
169
+ * teach (or poison) the shared cache. Default false (opt-in).
170
+ */
171
+ contributeToTeamCache: z4.boolean().default(false),
172
+ /**
173
+ * Roll infraspecific results up to the species: when an input resolves to an
174
+ * infraspecific accepted taxon (Variety / Subspecies / Form / …), replace it
175
+ * with its parent Species so every output row sits at species level.
176
+ *
177
+ * Default FALSE — the resolved rank is preserved as-is. Rolling up discards
178
+ * information the source data carried, so it is opt-in: turn it on only when
179
+ * the consuming system works at species level and you would otherwise have to
180
+ * flatten the export yourself.
181
+ */
182
+ speciesLevelAcceptedOnly: z4.boolean().default(false),
183
+ /**
184
+ * "The author isn't important." When the canonical name matches exactly one
185
+ * taxon but the input's author could not be CONFIRMED (flag
186
+ * `author-unconfirmed`), auto-accept the match instead of sending it to
187
+ * review. The taxon itself was never in doubt in that case — only whether
188
+ * the author string cites it the way WCVP does — so a dataset whose author
189
+ * column is unreliable (or absent from the source) can skip that queue.
190
+ *
191
+ * Default FALSE. This does NOT relax a genuine author CONFLICT
192
+ * (`author-mismatch`): a conflicting author may point at a different plant,
193
+ * so those still go to review. Every other review trigger (qualifiers,
194
+ * parse quality, force-review families, alternatives) is untouched.
195
+ */
196
+ acceptUnconfirmedAuthor: z4.boolean().default(false),
197
+ /**
198
+ * Which taxonomic backbone to match against. `wcvp` (default) runs the full
199
+ * cascade; `wfo` matches against the World Flora Online snapshot using the
200
+ * local layers (L1–L4). Stored on `jobs.referential`.
201
+ */
202
+ referential: z4.enum(["wcvp", "wfo"]).default("wcvp"),
203
+ /**
204
+ * Which snapshot version of the chosen backbone to match against. When
205
+ * omitted, the API resolves it at submit time: the team's configured default
206
+ * snapshot for this backbone (Admin → Backbone) when it is still installed,
207
+ * otherwise the most-recently-imported one. The job stores the resolved
208
+ * version on `jobs.referential_version` so re-runs are reproducible even if
209
+ * a newer snapshot lands later.
210
+ */
211
+ referentialVersion: z4.string().min(1).optional(),
91
212
  /**
92
- * Keep the identified accepted name at species level: when an input
93
- * resolves to an infraspecific accepted taxon (Variety / Subspecies /
94
- * Form / …), collapse it up to its parent Species. Default true; set false
95
- * to keep the exact infraspecific accepted taxon.
213
+ * WGSRPD Level-3 area code (e.g. 'MAS' = Massachusetts) the import is scoped
214
+ * to. When set, an ambiguous multi-candidate match is narrowed to the taxa
215
+ * that occur in this area — exactly one survivor resolves the match (Grade B,
216
+ * flagged `resolved-by-area:<code>`). null/omitted = no disambiguation.
96
217
  */
97
- speciesLevelAcceptedOnly: z3.boolean().default(true),
218
+ area: z4.string().nullable().optional(),
98
219
  /**
99
- * Which WCVP snapshot to match against. When omitted, the API picks the
100
- * most-recently-imported snapshot at submit time. The job stores the
101
- * resolved version on `jobs.referential_version` so re-runs are
102
- * reproducible even if a newer snapshot lands later.
220
+ * Auto-run the Pl@ntNet verification plugin for this job at completion —
221
+ * tags each match with whether its accepted taxon aligns to a Pl@ntNet
222
+ * species. Defaults on; set false to skip the auto-run for this job (it can
223
+ * still be triggered manually from the Plugins menu). Only has an effect
224
+ * when the server has Pl@ntNet configured (key set + not globally disabled);
225
+ * otherwise it is a no-op regardless.
103
226
  */
104
- referentialVersion: z3.string().min(1).optional(),
227
+ plantnet: z4.boolean().optional(),
105
228
  /**
106
229
  * Row filter: keep only rows whose `filterColumn` value equals
107
230
  * `filterValue` (trimmed, case-insensitive); all other rows are skipped at
@@ -109,306 +232,403 @@ var jobConfigSchema = z3.object({
109
232
  * apply. Lets a user match a subset of a mixed file (e.g. only
110
233
  * `kingdom = Plantae`).
111
234
  */
112
- filterColumn: z3.string().nullable().optional(),
113
- filterValue: z3.string().nullable().optional()
235
+ filterColumn: z4.string().nullable().optional(),
236
+ filterValue: z4.string().nullable().optional()
114
237
  });
115
- var wcvpSnapshotSummarySchema = z3.object({
116
- version: z3.string(),
117
- recordCount: z3.number().int().nonnegative(),
118
- importedAt: z3.string().datetime(),
119
- isLatest: z3.boolean()
120
- });
121
- var idTypeDetectionSchema = z3.object({
122
- sampleSize: z3.number().int().nonnegative(),
123
- counts: z3.record(z3.string(), z3.number().int().nonnegative()),
124
- dominant: z3.string().nullable(),
125
- dominantConfidence: z3.number().min(0).max(1),
126
- minorityExamples: z3.array(
127
- z3.object({
128
- rowIndex: z3.number().int().nonnegative(),
129
- value: z3.string(),
130
- type: z3.string()
238
+ var wcvpSnapshotSummarySchema = z4.object({
239
+ version: z4.string(),
240
+ recordCount: z4.number().int().nonnegative(),
241
+ importedAt: z4.string().datetime(),
242
+ /** Newest OFFICIAL snapshot (derived ones never count as latest). */
243
+ isLatest: z4.boolean(),
244
+ kind: z4.enum(["official", "derived"]),
245
+ baseVersion: z4.string().nullable(),
246
+ label: z4.string().nullable(),
247
+ ownerTeamId: z4.string().nullable()
248
+ });
249
+ var idTypeDetectionSchema = z4.object({
250
+ sampleSize: z4.number().int().nonnegative(),
251
+ counts: z4.record(z4.string(), z4.number().int().nonnegative()),
252
+ dominant: z4.string().nullable(),
253
+ dominantConfidence: z4.number().min(0).max(1),
254
+ minorityExamples: z4.array(
255
+ z4.object({
256
+ rowIndex: z4.number().int().nonnegative(),
257
+ value: z4.string(),
258
+ type: z4.string()
131
259
  })
132
260
  )
133
261
  });
134
- var publicAccessSchema = z3.enum(["none", "read", "review"]);
135
- var jobSummarySchema = z3.object({
136
- id: z3.string().ulid(),
137
- teamId: z3.string().uuid(),
138
- userId: z3.string().uuid(),
262
+ var publicAccessSchema = z4.enum(["none", "read", "review"]);
263
+ var jobSummarySchema = z4.object({
264
+ id: z4.string().ulid(),
265
+ teamId: z4.string().uuid(),
266
+ userId: z4.string().uuid(),
139
267
  /** Optional user-supplied label. URL still uses `id`; null when unset. */
140
- name: z3.string().nullable(),
141
- /** Display name of the user who submitted the job (from their token's
142
- * user). Null when unavailable (e.g. mutation responses). */
143
- submittedBy: z3.string().nullable(),
268
+ name: z4.string().nullable(),
269
+ /** The job's author: the label of the token that submitted it (snapshotted
270
+ * at creation, so it survives the token being renamed or revoked). Null for
271
+ * jobs created before authorship was tracked. */
272
+ submittedBy: z4.string().nullable(),
144
273
  status: jobStatusSchema,
145
274
  /** 'none' = private. 'read'/'review' = anyone with the link, no sign-in. */
146
275
  publicAccess: publicAccessSchema,
147
- referentialVersion: z3.string(),
148
- idColumn: z3.string().nullable(),
276
+ referentialVersion: z4.string(),
277
+ idColumn: z4.string().nullable(),
149
278
  idTypeDetection: idTypeDetectionSchema.nullable(),
150
- totalRows: z3.number().int().nonnegative(),
151
- uniqueQueries: z3.number().int().nonnegative(),
152
- processedRows: z3.number().int().nonnegative(),
153
- matchedRows: z3.number().int().nonnegative(),
154
- ambiguousRows: z3.number().int().nonnegative(),
155
- errorRows: z3.number().int().nonnegative(),
156
- needsReviewRows: z3.number().int().nonnegative(),
157
- llmCostCents: z3.number().int().nonnegative(),
158
- llmCostCapCents: z3.number().int().nonnegative(),
159
- createdAt: z3.string().datetime(),
160
- startedAt: z3.string().datetime().nullable(),
161
- completedAt: z3.string().datetime().nullable(),
162
- expiresAt: z3.string().datetime()
279
+ totalRows: z4.number().int().nonnegative(),
280
+ uniqueQueries: z4.number().int().nonnegative(),
281
+ processedRows: z4.number().int().nonnegative(),
282
+ matchedRows: z4.number().int().nonnegative(),
283
+ ambiguousRows: z4.number().int().nonnegative(),
284
+ errorRows: z4.number().int().nonnegative(),
285
+ needsReviewRows: z4.number().int().nonnegative(),
286
+ llmCostCents: z4.number().int().nonnegative(),
287
+ llmCostCapCents: z4.number().int().nonnegative(),
288
+ createdAt: z4.string().datetime(),
289
+ startedAt: z4.string().datetime().nullable(),
290
+ completedAt: z4.string().datetime().nullable(),
291
+ expiresAt: z4.string().datetime()
163
292
  });
164
- var jobPublicAccessUpdateSchema = z3.object({
293
+ var jobPublicAccessUpdateSchema = z4.object({
165
294
  publicAccess: publicAccessSchema
166
295
  });
167
- var progressEventSchema = z3.object({
168
- type: z3.enum(["progress", "status", "error", "completed"]),
169
- jobId: z3.string().ulid(),
170
- timestamp: z3.string().datetime(),
171
- processedRows: z3.number().int().nonnegative().optional(),
172
- totalRows: z3.number().int().nonnegative().optional(),
296
+ var jobRenameSchema = z4.object({
297
+ name: z4.string().trim().min(1).max(200)
298
+ });
299
+ var progressEventSchema = z4.object({
300
+ type: z4.enum(["progress", "status", "error", "completed"]),
301
+ jobId: z4.string().ulid(),
302
+ timestamp: z4.string().datetime(),
303
+ processedRows: z4.number().int().nonnegative().optional(),
304
+ totalRows: z4.number().int().nonnegative().optional(),
173
305
  status: jobStatusSchema.optional(),
174
- message: z3.string().optional()
306
+ message: z4.string().optional()
175
307
  });
176
- var taxonIdentifierRefSchema = z3.object({
177
- namespace: z3.string(),
178
- value: z3.string()
308
+ var taxonIdentifierRefSchema = z4.object({
309
+ namespace: z4.string(),
310
+ value: z4.string()
179
311
  });
180
- var jobRowSummarySchema = z3.object({
181
- id: z3.string().uuid(),
182
- rowIndex: z3.number().int().nonnegative(),
183
- inputName: z3.string().nullable(),
184
- inputId: z3.string().nullable(),
185
- inputFamily: z3.string().nullable(),
312
+ var jobRowSummarySchema = z4.object({
313
+ id: z4.string().uuid(),
314
+ rowIndex: z4.number().int().nonnegative(),
315
+ inputName: z4.string().nullable(),
316
+ inputId: z4.string().nullable(),
317
+ inputFamily: z4.string().nullable(),
186
318
  matchStatus: matchStatusSchema.nullable(),
187
319
  grade: gradeSchema.nullable(),
188
- evidenceType: z3.string().nullable(),
320
+ evidenceType: z4.string().nullable(),
189
321
  reviewStatus: reviewStatusSchema,
190
- matchQueryId: z3.string().uuid().nullable(),
191
- confidence: z3.number().nullable(),
192
- layer: z3.string().nullable(),
193
- flags: z3.array(z3.string()).nullable(),
194
- candidateAcceptedName: z3.string().nullable(),
322
+ matchQueryId: z4.string().uuid().nullable(),
323
+ confidence: z4.number().nullable(),
324
+ layer: z4.string().nullable(),
325
+ flags: z4.array(z4.string()).nullable(),
326
+ candidateAcceptedName: z4.string().nullable(),
195
327
  /** Authorship of the accepted taxon (distinct from `candidateAuthorship`
196
328
  * which carries the matched-row author when the match resolves via a synonym). */
197
- candidateAcceptedAuthorship: z3.string().nullable(),
198
- candidateAcceptedIdentifiers: z3.array(taxonIdentifierRefSchema).nullable(),
199
- candidateScientificName: z3.string().nullable(),
200
- candidateAuthorship: z3.string().nullable(),
201
- candidateFamily: z3.string().nullable(),
202
- candidateReason: z3.string().nullable(),
203
- candidateTargetUrl: z3.string().nullable()
204
- });
205
- var gradeCountsSchema = z3.object({
206
- A: z3.number().int().nonnegative(),
207
- B: z3.number().int().nonnegative(),
208
- C: z3.number().int().nonnegative(),
209
- ungraded: z3.number().int().nonnegative()
210
- });
211
- var statusCountsSchema = z3.object({
212
- matched: z3.number().int().nonnegative(),
213
- ambiguous: z3.number().int().nonnegative(),
214
- no_match: z3.number().int().nonnegative(),
215
- error: z3.number().int().nonnegative(),
216
- skipped: z3.number().int().nonnegative(),
329
+ candidateAcceptedAuthorship: z4.string().nullable(),
330
+ candidateAcceptedIdentifiers: z4.array(taxonIdentifierRefSchema).nullable(),
331
+ candidateScientificName: z4.string().nullable(),
332
+ candidateAuthorship: z4.string().nullable(),
333
+ candidateFamily: z4.string().nullable(),
334
+ candidateReason: z4.string().nullable(),
335
+ candidateTargetUrl: z4.string().nullable(),
336
+ /** Post-job verification-plugin results for this row, one per plugin that
337
+ * has a completed run. Empty when no plugin has run. Sourced from
338
+ * `job_plugin_results` joined via the row's match query. */
339
+ plugins: z4.array(jobRowPluginResultSchema).default([])
340
+ });
341
+ var gradeCountsSchema = z4.object({
342
+ A: z4.number().int().nonnegative(),
343
+ B: z4.number().int().nonnegative(),
344
+ C: z4.number().int().nonnegative(),
345
+ ungraded: z4.number().int().nonnegative()
346
+ });
347
+ var statusCountsSchema = z4.object({
348
+ matched: z4.number().int().nonnegative(),
349
+ ambiguous: z4.number().int().nonnegative(),
350
+ no_match: z4.number().int().nonnegative(),
351
+ error: z4.number().int().nonnegative(),
352
+ skipped: z4.number().int().nonnegative(),
217
353
  /** Rows whose match query exists but hasn't been processed yet (mid-run).
218
354
  * Distinct from `skipped` (no query — empty/guarded input). Not part of the
219
355
  * outcome funnel; the SPA can show it as in-progress. */
220
- pending: z3.number().int().nonnegative()
356
+ pending: z4.number().int().nonnegative()
221
357
  });
222
- var jobRowsPageSchema = z3.object({
223
- rows: z3.array(jobRowSummarySchema),
358
+ var layerCountsSchema = z4.record(z4.string(), z4.number().int().nonnegative());
359
+ var jobRowsPageSchema = z4.object({
360
+ rows: z4.array(jobRowSummarySchema),
224
361
  /** Row count matching the CURRENT filter (not the whole job) — drives
225
362
  * pagination so page count adapts to active grade/status/search/conf filters. */
226
- total: z3.number().int().nonnegative(),
227
- offset: z3.number().int().nonnegative(),
228
- limit: z3.number().int().positive(),
363
+ total: z4.number().int().nonnegative(),
364
+ offset: z4.number().int().nonnegative(),
365
+ limit: z4.number().int().positive(),
229
366
  /** Per-grade row counts for the whole job, independent of the current
230
367
  * filter. Used by the SPA to show "B (12)" next to each grade chip. */
231
368
  gradeCounts: gradeCountsSchema,
232
369
  /** Per-status row counts (matched / ambiguous / no_match / error /
233
370
  * skipped), same scope + purpose as `gradeCounts`. */
234
- statusCounts: statusCountsSchema
235
- });
236
- var wcvpSearchHitSchema = z3.object({
237
- taxonId: z3.string(),
238
- scientificName: z3.string(),
239
- canonicalName: z3.string(),
240
- authorship: z3.string().nullable(),
241
- rank: z3.string().nullable(),
242
- taxonomicStatus: z3.string().nullable(),
243
- family: z3.string().nullable(),
244
- acceptedTaxonId: z3.string().nullable(),
245
- acceptedName: z3.string().nullable(),
246
- similarity: z3.number()
247
- });
248
- var wcvpSearchResponseSchema = z3.array(wcvpSearchHitSchema);
249
- var jobDiffRowSchema = z3.object({
250
- rowId: z3.string().uuid(),
251
- rowIndex: z3.number().int().nonnegative(),
252
- inputName: z3.string().nullable(),
253
- inputFamily: z3.string().nullable(),
254
- matchedName: z3.string().nullable(),
255
- matchedAuthorship: z3.string().nullable(),
256
- matchedTaxonomicStatus: z3.string().nullable(),
257
- acceptedName: z3.string().nullable(),
258
- acceptedAuthorship: z3.string().nullable(),
259
- acceptedFamily: z3.string().nullable(),
260
- isSynonym: z3.boolean(),
261
- familyChanged: z3.boolean(),
371
+ statusCounts: statusCountsSchema,
372
+ /** Per-layer row counts, keyed by evidence type. Same scope + purpose as
373
+ * `gradeCounts`; only layers present in the job appear. */
374
+ layerCounts: layerCountsSchema
375
+ });
376
+ var wcvpRankLevelSchema = z4.enum(["genus", "species", "infra"]);
377
+ var wcvpSearchHitSchema = z4.object({
378
+ taxonId: z4.string(),
379
+ scientificName: z4.string(),
380
+ canonicalName: z4.string(),
381
+ authorship: z4.string().nullable(),
382
+ rank: z4.string().nullable(),
383
+ taxonomicStatus: z4.string().nullable(),
384
+ family: z4.string().nullable(),
385
+ acceptedTaxonId: z4.string().nullable(),
386
+ acceptedName: z4.string().nullable(),
387
+ similarity: z4.number()
388
+ });
389
+ var wcvpSearchResponseSchema = z4.array(wcvpSearchHitSchema);
390
+ var speciesRollupStatusSchema = z4.enum([
391
+ "available",
392
+ "already-species",
393
+ "no-species-ancestor",
394
+ "no-selection",
395
+ "unknown-taxon"
396
+ ]);
397
+ var speciesRollupTaxonSchema = z4.object({
398
+ taxonId: z4.string(),
399
+ scientificName: z4.string(),
400
+ canonicalName: z4.string(),
401
+ authorship: z4.string().nullable(),
402
+ rank: z4.string().nullable(),
403
+ family: z4.string().nullable()
404
+ });
405
+ var speciesRollupResponseSchema = z4.object({
406
+ status: speciesRollupStatusSchema,
407
+ /** The accepted taxon the row currently resolves to. */
408
+ current: speciesRollupTaxonSchema.nullable(),
409
+ /** Only set when `status` is `available`. */
410
+ target: speciesRollupTaxonSchema.nullable()
411
+ });
412
+ var jobDiffRowSchema = z4.object({
413
+ rowId: z4.string().uuid(),
414
+ rowIndex: z4.number().int().nonnegative(),
415
+ inputName: z4.string().nullable(),
416
+ inputFamily: z4.string().nullable(),
417
+ matchedName: z4.string().nullable(),
418
+ matchedAuthorship: z4.string().nullable(),
419
+ matchedTaxonomicStatus: z4.string().nullable(),
420
+ acceptedName: z4.string().nullable(),
421
+ acceptedAuthorship: z4.string().nullable(),
422
+ acceptedFamily: z4.string().nullable(),
423
+ isSynonym: z4.boolean(),
424
+ familyChanged: z4.boolean(),
262
425
  grade: gradeSchema.nullable(),
263
426
  reviewStatus: reviewStatusSchema,
264
- targetUrl: z3.string().nullable()
427
+ targetUrl: z4.string().nullable()
265
428
  });
266
- var jobDiffPageSchema = z3.object({
267
- rows: z3.array(jobDiffRowSchema),
268
- total: z3.number().int().nonnegative()
429
+ var jobDiffPageSchema = z4.object({
430
+ rows: z4.array(jobDiffRowSchema),
431
+ total: z4.number().int().nonnegative()
269
432
  });
270
- var reviewActionSchema = z3.enum(["accept", "reject"]);
271
- var reviewRequestSchema = z3.object({
433
+ var reviewActionSchema = z4.enum(["accept", "reject"]);
434
+ var reviewRequestSchema = z4.object({
272
435
  action: reviewActionSchema,
273
- candidateId: z3.string().uuid()
436
+ // Optional so a row with no candidates can still be resolved as "no
437
+ // match" (reject with no candidate). Accept always needs one.
438
+ candidateId: z4.string().uuid().optional()
439
+ }).refine((v) => v.action === "reject" || !!v.candidateId, {
440
+ message: "candidateId is required to accept a candidate",
441
+ path: ["candidateId"]
274
442
  });
275
- var bulkReviewRequestSchema = z3.object({
443
+ var bulkReviewRequestSchema = z4.object({
276
444
  action: reviewActionSchema,
277
- filter: z3.object({
445
+ filter: z4.object({
278
446
  grade: gradeSchema.optional(),
279
447
  matchStatus: matchStatusSchema.optional(),
280
- family: z3.string().optional()
448
+ family: z4.string().optional(),
449
+ /** Restrict the bulk action to these specific match-query ids. The
450
+ * review queue's grouped/batch view computes a group's membership
451
+ * client-side (by issue category or author-equivalence pair) and
452
+ * sends the exact ids — flag/category logic that a coarse
453
+ * grade/family filter can't express. ANDed with any other filter;
454
+ * the server still restricts to currently-pending, owned queries. */
455
+ matchQueryIds: z4.array(z4.string().uuid()).max(5e4).optional()
281
456
  }).default({}),
282
- dryRun: z3.boolean().default(false)
457
+ dryRun: z4.boolean().default(false)
283
458
  });
284
- var bulkReviewResponseSchema = z3.object({
459
+ var bulkReviewResponseSchema = z4.object({
285
460
  action: reviewActionSchema,
286
- matchedQueries: z3.number().int().nonnegative(),
287
- dryRun: z3.boolean(),
288
- appliedAt: z3.string().datetime().nullable()
461
+ matchedQueries: z4.number().int().nonnegative(),
462
+ dryRun: z4.boolean(),
463
+ appliedAt: z4.string().datetime().nullable()
289
464
  });
290
- var reviewResponseSchema = z3.object({
291
- matchQueryId: z3.string().uuid(),
465
+ var reviewResponseSchema = z4.object({
466
+ matchQueryId: z4.string().uuid(),
292
467
  action: reviewActionSchema,
293
- candidateId: z3.string().uuid(),
468
+ // Null only for a "no match" reject (a row with no candidate). An accept
469
+ // always resolves to a candidate — the refine catches a server that
470
+ // returned an accept without one.
471
+ candidateId: z4.string().uuid().nullable(),
294
472
  reviewStatus: reviewStatusSchema,
295
- reviewEventId: z3.string().uuid()
473
+ reviewEventId: z4.string().uuid()
474
+ }).refine((v) => v.action !== "accept" || v.candidateId !== null, {
475
+ message: "an accept must resolve to a candidateId",
476
+ path: ["candidateId"]
296
477
  });
297
- var matchAttemptSummarySchema = z3.object({
298
- id: z3.string().uuid(),
299
- layer: z3.string(),
300
- provider: z3.string(),
301
- status: z3.string(),
302
- durationMs: z3.number().int().nonnegative(),
478
+ var matchAttemptSummarySchema = z4.object({
479
+ id: z4.string().uuid(),
480
+ layer: z4.string(),
481
+ provider: z4.string(),
482
+ status: z4.string(),
483
+ durationMs: z4.number().int().nonnegative(),
303
484
  // JSONB fields accept any shape. Note: `z.unknown()` infers as optional,
304
485
  // so consumers should guard for `undefined` even though we always send
305
486
  // them as part of the response.
306
- query: z3.unknown(),
307
- rawResponseSummary: z3.unknown().nullable(),
308
- createdAt: z3.string().datetime()
487
+ query: z4.unknown(),
488
+ rawResponseSummary: z4.unknown().nullable(),
489
+ createdAt: z4.string().datetime()
309
490
  });
310
- var matchCandidateSummarySchema = z3.object({
311
- id: z3.string().uuid(),
312
- attemptId: z3.string().uuid(),
313
- source: z3.string(),
314
- sourceId: z3.string().nullable(),
315
- scientificName: z3.string(),
316
- canonicalName: z3.string(),
317
- authorship: z3.string().nullable(),
318
- rank: z3.string().nullable(),
319
- taxonomicStatus: z3.string().nullable(),
320
- acceptedName: z3.string().nullable(),
321
- acceptedAuthorship: z3.string().nullable(),
322
- acceptedIdentifiers: z3.array(taxonIdentifierRefSchema).nullable(),
323
- identifiers: z3.array(taxonIdentifierRefSchema),
324
- family: z3.string().nullable(),
325
- confidence: z3.number().nullable(),
491
+ var matchCandidateSummarySchema = z4.object({
492
+ id: z4.string().uuid(),
493
+ attemptId: z4.string().uuid(),
494
+ source: z4.string(),
495
+ sourceId: z4.string().nullable(),
496
+ scientificName: z4.string(),
497
+ canonicalName: z4.string(),
498
+ authorship: z4.string().nullable(),
499
+ rank: z4.string().nullable(),
500
+ taxonomicStatus: z4.string().nullable(),
501
+ acceptedName: z4.string().nullable(),
502
+ acceptedAuthorship: z4.string().nullable(),
503
+ acceptedIdentifiers: z4.array(taxonIdentifierRefSchema).nullable(),
504
+ identifiers: z4.array(taxonIdentifierRefSchema),
505
+ family: z4.string().nullable(),
506
+ confidence: z4.number().nullable(),
326
507
  grade: gradeSchema.nullable(),
327
- reason: z3.string().nullable(),
328
- flags: z3.array(z3.string()).nullable(),
329
- state: z3.enum(["proposed", "accepted", "rejected", "overridden"]),
330
- targetUrl: z3.string().nullable()
508
+ reason: z4.string().nullable(),
509
+ flags: z4.array(z4.string()).nullable(),
510
+ state: z4.enum(["proposed", "accepted", "rejected", "overridden"]),
511
+ targetUrl: z4.string().nullable()
331
512
  });
332
- var matchQueryDetailSchema = z3.object({
333
- id: z3.string().uuid(),
334
- jobId: z3.string().ulid(),
335
- normalizedInput: z3.string(),
336
- parsed: z3.unknown(),
513
+ var matchQueryDetailSchema = z4.object({
514
+ id: z4.string().uuid(),
515
+ jobId: z4.string().ulid(),
516
+ normalizedInput: z4.string(),
517
+ parsed: z4.unknown(),
337
518
  matchStatus: matchStatusSchema.nullable(),
338
- evidenceType: z3.string().nullable(),
519
+ evidenceType: z4.string().nullable(),
339
520
  grade: gradeSchema.nullable(),
340
- confidence: z3.number().nullable(),
341
- flags: z3.array(z3.string()).nullable(),
521
+ confidence: z4.number().nullable(),
522
+ flags: z4.array(z4.string()).nullable(),
342
523
  reviewStatus: reviewStatusSchema,
343
- selectedCandidateId: z3.string().uuid().nullable()
524
+ selectedCandidateId: z4.string().uuid().nullable()
525
+ });
526
+ var otherVersionMatchSchema = z4.object({
527
+ version: z4.string(),
528
+ importedAt: z4.string(),
529
+ isLatest: z4.boolean(),
530
+ scientificName: z4.string(),
531
+ authorship: z4.string().nullable(),
532
+ taxonomicStatus: z4.string().nullable(),
533
+ acceptedName: z4.string().nullable()
344
534
  });
345
- var queryCandidatesResponseSchema = z3.object({
535
+ var queryCandidatesResponseSchema = z4.object({
346
536
  query: matchQueryDetailSchema,
347
- candidates: z3.array(matchCandidateSummarySchema),
348
- attempts: z3.array(matchAttemptSummarySchema)
349
- });
350
- var reviewChatMessageSchema = z3.object({
351
- role: z3.enum(["user", "assistant"]),
352
- content: z3.string().min(1).max(4e3)
353
- });
354
- var reviewChatBodySchema = z3.object({
355
- messages: z3.array(reviewChatMessageSchema).min(1).max(40)
356
- });
357
- var reviewChatResponseSchema = z3.object({
358
- reply: z3.string(),
359
- model: z3.string()
360
- });
361
- var replayableLayerSchema = z3.enum(["L1", "L2", "L3", "L4", "L5", "L6", "L7"]);
362
- var replayLayerBodySchema = z3.object({ layer: replayableLayerSchema });
363
- var replayCandidateSchema = z3.object({
364
- scientificName: z3.string(),
365
- canonicalName: z3.string(),
366
- authorship: z3.string().nullable(),
367
- rank: z3.string().nullable(),
368
- taxonomicStatus: z3.string().nullable(),
369
- acceptedName: z3.string().nullable(),
370
- acceptedAuthorship: z3.string().nullable(),
371
- family: z3.string().nullable(),
537
+ candidates: z4.array(matchCandidateSummarySchema),
538
+ attempts: z4.array(matchAttemptSummarySchema),
539
+ /** A newer/other backbone version that has this exact name (see schema). */
540
+ otherVersionMatch: otherVersionMatchSchema.nullable()
541
+ });
542
+ var reviewChatMessageSchema = z4.object({
543
+ role: z4.enum(["user", "assistant"]),
544
+ content: z4.string().min(1).max(4e3)
545
+ });
546
+ var reviewChatBodySchema = z4.object({
547
+ messages: z4.array(reviewChatMessageSchema).min(1).max(40),
548
+ /** Let the assistant also search the web (OpenRouter web plugin). Default on
549
+ * server-side when omitted; the client sends it explicitly from its toggle. */
550
+ webSearch: z4.boolean().optional(),
551
+ /** Per-chat model override (an OpenRouter id / alias). When omitted the
552
+ * team's configured heavy model is used. Length-capped defensively. */
553
+ model: z4.string().trim().min(1).max(200).optional()
554
+ });
555
+ var reviewChatResponseSchema = z4.object({
556
+ reply: z4.string(),
557
+ model: z4.string()
558
+ });
559
+ var reviewChatStreamEventSchema = z4.discriminatedUnion("type", [
560
+ /** A chunk of the model's visible answer. */
561
+ z4.object({ type: z4.literal("text"), delta: z4.string() }),
562
+ /** A chunk of the model's reasoning/thinking (when the model exposes it). */
563
+ z4.object({ type: z4.literal("reasoning"), delta: z4.string() }),
564
+ /** The model decided to call a tool, with the (JSON) arguments it chose. */
565
+ z4.object({ type: z4.literal("tool_call"), id: z4.string(), name: z4.string(), arguments: z4.string() }),
566
+ /** The result we fed back to the model after running that tool. `sql` is the
567
+ * statement the tool actually ran, shown to the reviewer for transparency —
568
+ * it is emitted on this event ONLY and never added to the model's context. */
569
+ z4.object({
570
+ type: z4.literal("tool_result"),
571
+ id: z4.string(),
572
+ name: z4.string(),
573
+ result: z4.string(),
574
+ sql: z4.string().optional()
575
+ }),
576
+ /** Terminal success — carries the model id that answered. */
577
+ z4.object({ type: z4.literal("done"), model: z4.string() }),
578
+ /** Terminal failure — a human-readable reason. */
579
+ z4.object({ type: z4.literal("error"), error: z4.string() })
580
+ ]);
581
+ var replayableLayerSchema = z4.enum(["L1", "L2", "L3", "L4", "L5", "L6", "L7"]);
582
+ var replayLayerBodySchema = z4.object({ layer: replayableLayerSchema });
583
+ var replayCandidateSchema = z4.object({
584
+ scientificName: z4.string(),
585
+ canonicalName: z4.string(),
586
+ authorship: z4.string().nullable(),
587
+ rank: z4.string().nullable(),
588
+ taxonomicStatus: z4.string().nullable(),
589
+ acceptedName: z4.string().nullable(),
590
+ acceptedAuthorship: z4.string().nullable(),
591
+ family: z4.string().nullable(),
372
592
  /** WCVP taxon id of the matched row (null for unresolved external proposals). */
373
- taxonId: z3.string().nullable(),
593
+ taxonId: z4.string().nullable(),
374
594
  /** WCVP taxon id of the resolved accepted taxon (null when unresolved). */
375
- acceptedTaxonId: z3.string().nullable(),
376
- confidence: z3.number().nullable(),
377
- reason: z3.string().nullable(),
378
- flags: z3.array(z3.string()),
379
- targetUrl: z3.string().nullable(),
595
+ acceptedTaxonId: z4.string().nullable(),
596
+ confidence: z4.number().nullable(),
597
+ reason: z4.string().nullable(),
598
+ flags: z4.array(z4.string()),
599
+ targetUrl: z4.string().nullable(),
380
600
  /** Provider that proposed this candidate (e.g. tnrs, gbif, openrouter,
381
601
  * wcvp-local). */
382
- provider: z3.string().nullable()
602
+ provider: z4.string().nullable()
383
603
  });
384
- var replayAttemptSchema = z3.object({
385
- provider: z3.string(),
386
- status: z3.string(),
387
- durationMs: z3.number().int().nonnegative(),
388
- summary: z3.unknown().nullable()
604
+ var replayAttemptSchema = z4.object({
605
+ provider: z4.string(),
606
+ status: z4.string(),
607
+ durationMs: z4.number().int().nonnegative(),
608
+ summary: z4.unknown().nullable()
389
609
  });
390
- var replayLayerResultSchema = z3.object({
610
+ var replayLayerResultSchema = z4.object({
391
611
  layer: replayableLayerSchema,
392
- kind: z3.enum(["unique", "ambiguous", "miss"]),
393
- reason: z3.string().nullable(),
612
+ kind: z4.enum(["unique", "ambiguous", "miss"]),
613
+ reason: z4.string().nullable(),
394
614
  /** Human-facing note when the layer couldn't run as-is (e.g. LLM not
395
615
  * configured, no external providers enabled). */
396
- note: z3.string().nullable(),
397
- providers: z3.array(z3.string()),
398
- durationMs: z3.number().int().nonnegative(),
616
+ note: z4.string().nullable(),
617
+ providers: z4.array(z4.string()),
618
+ durationMs: z4.number().int().nonnegative(),
399
619
  /** What was actually submitted to the layer (post gnparser). */
400
- query: z3.object({
401
- canonical: z3.string(),
402
- authorship: z3.string().nullable(),
403
- inputId: z3.string().nullable()
620
+ query: z4.object({
621
+ canonical: z4.string(),
622
+ authorship: z4.string().nullable(),
623
+ inputId: z4.string().nullable()
404
624
  }),
405
- candidates: z3.array(replayCandidateSchema),
406
- attempts: z3.array(replayAttemptSchema)
625
+ candidates: z4.array(replayCandidateSchema),
626
+ attempts: z4.array(replayAttemptSchema)
407
627
  });
408
628
 
409
629
  // ../shared/src/schemas/auth.ts
410
- import { z as z4 } from "zod";
411
- var tokenScopeSchema = z4.enum([
630
+ import { z as z5 } from "zod";
631
+ var tokenScopeSchema = z5.enum([
412
632
  "submit:job",
413
633
  "read:job",
414
634
  "cancel:job",
@@ -420,77 +640,83 @@ var tokenScopeSchema = z4.enum([
420
640
  "admin:teams"
421
641
  ]);
422
642
  var ALL_TOKEN_SCOPES = tokenScopeSchema.options;
423
- var tokenStringSchema = z4.string().regex(/^ptm_[a-zA-Z0-9]{12}_[a-zA-Z0-9]{32}$/, "Malformed token");
424
- var meSchema = z4.object({
425
- userId: z4.string().uuid(),
426
- teamId: z4.string().uuid(),
427
- displayName: z4.string(),
428
- scopes: z4.array(tokenScopeSchema),
429
- tokenLabel: z4.string().nullable()
643
+ var tokenStringSchema = z5.string().regex(/^ptm_[a-zA-Z0-9]{12}_[a-zA-Z0-9]{32}$/, "Malformed token");
644
+ var meSchema = z5.object({
645
+ userId: z5.string().uuid(),
646
+ teamId: z5.string().uuid(),
647
+ displayName: z5.string(),
648
+ scopes: z5.array(tokenScopeSchema),
649
+ tokenLabel: z5.string().nullable(),
650
+ /** DB id of the token that authenticated this request — matches a row's
651
+ * `id` in the admin token list, so the UI can highlight "this is you". */
652
+ tokenId: z5.string().uuid().nullable()
430
653
  });
431
- var tokenCreateBodySchema = z4.object({
432
- label: z4.string().min(1).max(120),
433
- scopes: z4.array(tokenScopeSchema).min(1),
434
- expiresAt: z4.string().datetime().optional()
654
+ var tokenCreateBodySchema = z5.object({
655
+ label: z5.string().min(1).max(120),
656
+ scopes: z5.array(tokenScopeSchema).min(1),
657
+ expiresAt: z5.string().datetime().optional()
435
658
  });
436
- var tokenSummarySchema = z4.object({
437
- id: z4.string().uuid(),
438
- tokenIdPrefix: z4.string(),
439
- label: z4.string(),
440
- displayName: z4.string(),
441
- scopes: z4.array(tokenScopeSchema),
442
- createdAt: z4.string().datetime(),
443
- lastUsedAt: z4.string().datetime().nullable(),
444
- expiresAt: z4.string().datetime().nullable()
659
+ var tokenRenameSchema = z5.object({
660
+ label: z5.string().trim().min(1).max(120)
661
+ });
662
+ var tokenSummarySchema = z5.object({
663
+ id: z5.string().uuid(),
664
+ tokenIdPrefix: z5.string(),
665
+ label: z5.string(),
666
+ displayName: z5.string(),
667
+ scopes: z5.array(tokenScopeSchema),
668
+ createdAt: z5.string().datetime(),
669
+ lastUsedAt: z5.string().datetime().nullable(),
670
+ expiresAt: z5.string().datetime().nullable()
445
671
  });
446
672
  var tokenCreatedSchema = tokenSummarySchema.extend({
447
673
  tokenString: tokenStringSchema
448
674
  });
449
- var teamSummarySchema = z4.object({
450
- id: z4.string().uuid(),
451
- name: z4.string(),
452
- createdAt: z4.string().datetime()
675
+ var teamSummarySchema = z5.object({
676
+ id: z5.string().uuid(),
677
+ name: z5.string(),
678
+ createdAt: z5.string().datetime()
453
679
  });
454
- var teamCreateBodySchema = z4.object({
455
- name: z4.string().min(1).max(120),
456
- initialUserDisplayName: z4.string().min(1).max(120).default("Admin"),
457
- initialToken: z4.object({
458
- label: z4.string().min(1).max(120),
459
- scopes: z4.array(tokenScopeSchema).min(1)
680
+ var teamCreateBodySchema = z5.object({
681
+ name: z5.string().min(1).max(120),
682
+ initialUserDisplayName: z5.string().min(1).max(120).default("Admin"),
683
+ initialToken: z5.object({
684
+ label: z5.string().min(1).max(120),
685
+ scopes: z5.array(tokenScopeSchema).min(1)
460
686
  }).optional()
461
687
  });
462
688
  var teamCreatedSchema = teamSummarySchema.extend({
463
- initialUser: z4.object({
464
- id: z4.string().uuid(),
465
- displayName: z4.string()
689
+ initialUser: z5.object({
690
+ id: z5.string().uuid(),
691
+ displayName: z5.string()
466
692
  }),
467
693
  initialToken: tokenCreatedSchema.nullable()
468
694
  });
469
- var setupStatusSchema = z4.object({
470
- needsSetup: z4.boolean()
695
+ var setupStatusSchema = z5.object({
696
+ needsSetup: z5.boolean()
471
697
  });
472
- var setupBodySchema = z4.object({
473
- teamName: z4.string().min(1).max(120),
474
- displayName: z4.string().min(1).max(120).default("Admin")
698
+ var setupBodySchema = z5.object({
699
+ teamName: z5.string().min(1).max(120),
700
+ displayName: z5.string().min(1).max(120).default("Admin")
475
701
  });
476
- var setupResultSchema = z4.object({
702
+ var setupResultSchema = z5.object({
477
703
  team: teamSummarySchema,
478
- user: z4.object({ id: z4.string().uuid(), displayName: z4.string() }),
704
+ user: z5.object({ id: z5.string().uuid(), displayName: z5.string() }),
479
705
  tokenString: tokenStringSchema,
480
- scopes: z4.array(tokenScopeSchema)
706
+ scopes: z5.array(tokenScopeSchema)
481
707
  });
482
- var teamLlmSettingsSchema = z4.object({
483
- keyConfigured: z4.boolean(),
484
- keyHint: z4.string().nullable(),
485
- lightModel: z4.string(),
486
- heavyModel: z4.string()
708
+ var teamLlmSettingsSchema = z5.object({
709
+ keyConfigured: z5.boolean(),
710
+ keyHint: z5.string().nullable(),
711
+ lightModel: z5.string(),
712
+ heavyModel: z5.string()
487
713
  });
488
- var teamLlmUpdateSchema = z4.object({
489
- openrouterApiKey: z4.string().max(400).nullable().optional(),
490
- lightModel: z4.string().min(1).max(200).optional(),
491
- heavyModel: z4.string().min(1).max(200).optional()
714
+ var teamLlmUpdateSchema = z5.object({
715
+ openrouterApiKey: z5.string().max(400).nullable().optional(),
716
+ lightModel: z5.string().min(1).max(200).optional(),
717
+ heavyModel: z5.string().min(1).max(200).optional()
492
718
  });
493
- var wcvpImportStatusSchema = z4.enum([
719
+ var wcvpImportStatusSchema = z5.enum([
494
720
  "queued",
495
721
  "downloading",
496
722
  "importing",
@@ -498,68 +724,68 @@ var wcvpImportStatusSchema = z4.enum([
498
724
  "failed",
499
725
  "cancelled"
500
726
  ]);
501
- var wcvpImportRunSchema = z4.object({
502
- id: z4.string().uuid(),
503
- version: z4.string(),
504
- sourceUrl: z4.string(),
727
+ var wcvpImportRunSchema = z5.object({
728
+ id: z5.string().uuid(),
729
+ version: z5.string(),
730
+ sourceUrl: z5.string(),
505
731
  status: wcvpImportStatusSchema,
506
- forceOverwrite: z4.boolean(),
507
- bytesDownloaded: z4.number().int().nonnegative(),
508
- totalBytes: z4.number().int().nonnegative().nullable(),
509
- insertedCount: z4.number().int().nonnegative(),
510
- recordCount: z4.number().int().nonnegative().nullable(),
511
- message: z4.string().nullable(),
512
- createdAt: z4.string().datetime(),
513
- startedAt: z4.string().datetime().nullable(),
514
- completedAt: z4.string().datetime().nullable()
732
+ forceOverwrite: z5.boolean(),
733
+ bytesDownloaded: z5.number().int().nonnegative(),
734
+ totalBytes: z5.number().int().nonnegative().nullable(),
735
+ insertedCount: z5.number().int().nonnegative(),
736
+ recordCount: z5.number().int().nonnegative().nullable(),
737
+ message: z5.string().nullable(),
738
+ createdAt: z5.string().datetime(),
739
+ startedAt: z5.string().datetime().nullable(),
740
+ completedAt: z5.string().datetime().nullable()
515
741
  });
516
- var wcvpImportCreateBodySchema = z4.object({
517
- version: z4.string().min(1).max(120),
742
+ var wcvpImportCreateBodySchema = z5.object({
743
+ version: z5.string().min(1).max(120),
518
744
  /** Defaults to the Kew SFTP URL when omitted, so the admin doesn't have
519
745
  * to remember it for routine v13/v14 imports. */
520
- url: z4.string().url().optional(),
521
- force: z4.boolean().optional()
746
+ url: z5.string().url().optional(),
747
+ force: z5.boolean().optional()
522
748
  });
523
- var adminHealthErrorSchema = z4.object({
524
- id: z4.string().uuid(),
525
- action: z4.string(),
526
- entityType: z4.string(),
527
- entityId: z4.string().nullable(),
528
- timestamp: z4.string().datetime(),
529
- metadata: z4.unknown().nullable()
530
- });
531
- var adminHealthSchema = z4.object({
532
- queueDepth: z4.object({
533
- waiting: z4.number().int().nonnegative(),
534
- active: z4.number().int().nonnegative(),
535
- delayed: z4.number().int().nonnegative(),
536
- completed: z4.number().int().nonnegative(),
537
- failed: z4.number().int().nonnegative(),
538
- paused: z4.number().int().nonnegative()
749
+ var adminHealthErrorSchema = z5.object({
750
+ id: z5.string().uuid(),
751
+ action: z5.string(),
752
+ entityType: z5.string(),
753
+ entityId: z5.string().nullable(),
754
+ timestamp: z5.string().datetime(),
755
+ metadata: z5.unknown().nullable()
756
+ });
757
+ var adminHealthSchema = z5.object({
758
+ queueDepth: z5.object({
759
+ waiting: z5.number().int().nonnegative(),
760
+ active: z5.number().int().nonnegative(),
761
+ delayed: z5.number().int().nonnegative(),
762
+ completed: z5.number().int().nonnegative(),
763
+ failed: z5.number().int().nonnegative(),
764
+ paused: z5.number().int().nonnegative()
539
765
  }),
540
- workerCount: z4.number().int().nonnegative(),
541
- lastErrors: z4.array(adminHealthErrorSchema),
542
- rateLimitBudget: z4.object({
543
- max: z4.number().int().nonnegative(),
544
- timeWindowSeconds: z4.number().int().nonnegative()
766
+ workerCount: z5.number().int().nonnegative(),
767
+ lastErrors: z5.array(adminHealthErrorSchema),
768
+ rateLimitBudget: z5.object({
769
+ max: z5.number().int().nonnegative(),
770
+ timeWindowSeconds: z5.number().int().nonnegative()
545
771
  }),
546
- snapshotVersion: z4.string().nullable(),
547
- snapshotImportedAt: z4.string().datetime().nullable(),
772
+ snapshotVersion: z5.string().nullable(),
773
+ snapshotImportedAt: z5.string().datetime().nullable(),
548
774
  /**
549
775
  * Number of cached accepted-name mappings (L0.5 match-history rows) for the
550
776
  * caller's team — the size of the team accepted-name cache.
551
777
  */
552
- teamHistorySize: z4.number().int().nonnegative(),
778
+ teamHistorySize: z5.number().int().nonnegative(),
553
779
  /**
554
780
  * Overall (all-teams) tally of which cascade layer produced the winning
555
781
  * candidate, across every matched query. `layer` is the raw attempt layer
556
782
  * (`L1`…`L7`, `L0.5`, `OVERRIDE`); `count` is the number of matched queries
557
783
  * that layer resolved. Descending by count.
558
784
  */
559
- layerUsage: z4.array(
560
- z4.object({
561
- layer: z4.string(),
562
- count: z4.number().int().nonnegative()
785
+ layerUsage: z5.array(
786
+ z5.object({
787
+ layer: z5.string(),
788
+ count: z5.number().int().nonnegative()
563
789
  })
564
790
  ),
565
791
  /**
@@ -569,18 +795,239 @@ var adminHealthSchema = z4.object({
569
795
  * Lets the health page surface each external source's usage, hit/miss/error
570
796
  * breakdown, latency, and recency — so a misbehaving provider is visible.
571
797
  */
572
- externalProviders: z4.array(
573
- z4.object({
574
- provider: z4.string(),
575
- total: z4.number().int().nonnegative(),
576
- hits: z4.number().int().nonnegative(),
577
- misses: z4.number().int().nonnegative(),
578
- errors: z4.number().int().nonnegative(),
579
- avgDurationMs: z4.number().int().nonnegative(),
580
- lastUsedAt: z4.string().datetime().nullable()
798
+ externalProviders: z5.array(
799
+ z5.object({
800
+ provider: z5.string(),
801
+ total: z5.number().int().nonnegative(),
802
+ hits: z5.number().int().nonnegative(),
803
+ misses: z5.number().int().nonnegative(),
804
+ errors: z5.number().int().nonnegative(),
805
+ avgDurationMs: z5.number().int().nonnegative(),
806
+ lastUsedAt: z5.string().datetime().nullable()
807
+ })
808
+ )
809
+ });
810
+ var matchHistoryEntrySchema = z5.object({
811
+ id: z5.string().uuid(),
812
+ /** The normalized input name this mapping is keyed on. */
813
+ normalizedInput: z5.string(),
814
+ acceptedName: z5.string(),
815
+ acceptedIdentifier: z5.object({ namespace: z5.string(), value: z5.string() }).nullable(),
816
+ /** Resolved from the backbone snapshot at read time (null when the stored
817
+ * identifier no longer resolves): authorship + family of the accepted taxon,
818
+ * its identifiers (backbone id, IPNI LSID, portal url) and the portal link. */
819
+ acceptedAuthorship: z5.string().nullable(),
820
+ family: z5.string().nullable(),
821
+ acceptedIdentifiers: z5.array(z5.object({ namespace: z5.string(), value: z5.string() })),
822
+ targetUrl: z5.string().nullable(),
823
+ targetReferential: z5.string(),
824
+ referentialVersion: z5.string(),
825
+ independentUserAcceptanceCount: z5.number().int().nonnegative(),
826
+ rejectionCount: z5.number().int().nonnegative(),
827
+ blacklisted: z5.boolean(),
828
+ needsReconfirmation: z5.boolean(),
829
+ /** Currently usable as a cache hit (not blacklisted, acceptances > rejections). */
830
+ active: z5.boolean(),
831
+ /** Auto-accept (Grade A): acceptances ≥ the team's promotion threshold. */
832
+ promoted: z5.boolean(),
833
+ /** Label of the token that last reviewed this mapping (falls back to the
834
+ * reviewer's display name for rows predating token tracking). */
835
+ lastReviewedBy: z5.string().nullable(),
836
+ firstSeenAt: z5.string().datetime(),
837
+ lastAcceptedAt: z5.string().datetime().nullable()
838
+ });
839
+ var teamCacheResponseSchema = z5.object({
840
+ /** Distinct-user acceptances needed to promote a hit to Grade A. */
841
+ promotionThreshold: z5.number().int().positive(),
842
+ /** Total entries in the team cache (before the limit). */
843
+ total: z5.number().int().nonnegative(),
844
+ entries: z5.array(matchHistoryEntrySchema)
845
+ });
846
+
847
+ // ../shared/src/schemas/backbone.ts
848
+ import { z as z6 } from "zod";
849
+ var backboneSchema = z6.enum(["wcvp", "wfo"]);
850
+ var wfoImportStatusSchema = z6.enum([
851
+ "queued",
852
+ "downloading",
853
+ "importing",
854
+ "completed",
855
+ "failed",
856
+ "cancelled"
857
+ ]);
858
+ var wfoImportRunSchema = z6.object({
859
+ id: z6.string().uuid(),
860
+ version: z6.string(),
861
+ sourceUrl: z6.string(),
862
+ status: wfoImportStatusSchema,
863
+ forceOverwrite: z6.boolean(),
864
+ bytesDownloaded: z6.number().int().nonnegative(),
865
+ totalBytes: z6.number().int().nonnegative().nullable(),
866
+ insertedCount: z6.number().int().nonnegative(),
867
+ recordCount: z6.number().int().nonnegative().nullable(),
868
+ message: z6.string().nullable(),
869
+ createdAt: z6.string().datetime(),
870
+ startedAt: z6.string().datetime().nullable(),
871
+ completedAt: z6.string().datetime().nullable()
872
+ });
873
+ var wfoImportCreateBodySchema = z6.object({
874
+ version: z6.string().min(1).max(120),
875
+ /** Direct download URL (from the Zenodo version listing). Required for WFO
876
+ * since versions aren't derivable from the label like WCVP's Kew URLs. */
877
+ url: z6.string().url().optional(),
878
+ force: z6.boolean().optional()
879
+ });
880
+ var snapshotKindSchema = z6.enum(["official", "derived"]);
881
+ var wfoSnapshotSummarySchema = z6.object({
882
+ version: z6.string(),
883
+ recordCount: z6.number().int().nonnegative(),
884
+ importedAt: z6.string().datetime(),
885
+ isLatest: z6.boolean(),
886
+ kind: snapshotKindSchema,
887
+ baseVersion: z6.string().nullable(),
888
+ label: z6.string().nullable(),
889
+ ownerTeamId: z6.string().nullable()
890
+ });
891
+ var wfoVersionSchema = z6.object({
892
+ /** Release label, e.g. "2025-12". */
893
+ version: z6.string(),
894
+ /** Zenodo record id for this version. */
895
+ recordId: z6.string(),
896
+ /** The plant-list archive filename inside the record. */
897
+ fileName: z6.string(),
898
+ /** Direct content download URL. */
899
+ url: z6.string().url(),
900
+ sizeBytes: z6.number().int().nonnegative().nullable(),
901
+ publishedAt: z6.string().nullable(),
902
+ isLatest: z6.boolean()
903
+ });
904
+ var wfoVersionListSchema = z6.array(wfoVersionSchema);
905
+ var backboneKeySchema = z6.enum(["wcvp", "wfo"]);
906
+ var backboneDefaultsSchema = z6.object({
907
+ wcvp: z6.string().nullable(),
908
+ wfo: z6.string().nullable()
909
+ });
910
+ var backboneDefaultUpdateSchema = z6.object({
911
+ backbone: backboneKeySchema,
912
+ version: z6.string().min(1).nullable()
913
+ });
914
+ var DERIVED_SLUG_RE = /^[a-z0-9][a-z0-9-]{1,40}$/;
915
+ var MAX_BASE_VERSION_LENGTH = 120;
916
+ var MAX_SNAPSHOT_VERSION_LENGTH = MAX_BASE_VERSION_LENGTH + 1 + 41;
917
+ var derivedUploadMetaSchema = z6.object({
918
+ baseVersion: z6.string().min(1).max(MAX_BASE_VERSION_LENGTH),
919
+ slug: z6.string().regex(DERIVED_SLUG_RE, "slug: lowercase letters, digits and dashes (2\u201341 chars)"),
920
+ label: z6.string().trim().min(1).max(120),
921
+ notes: z6.string().trim().max(2e3).optional()
922
+ });
923
+ var derivedUploadStatusSchema = z6.enum([
924
+ "staging",
925
+ "ready",
926
+ "invalid",
927
+ "importing",
928
+ "imported",
929
+ "failed",
930
+ "discarded"
931
+ ]);
932
+ var derivedIssueSeveritySchema = z6.enum(["error", "warning"]);
933
+ var derivedIssueSchema = z6.object({
934
+ code: z6.string(),
935
+ severity: derivedIssueSeveritySchema,
936
+ count: z6.number().int().nonnegative(),
937
+ /** Free-form detail for the UI (e.g. the offending vocabulary values). */
938
+ detail: z6.string().nullable(),
939
+ samples: z6.array(
940
+ z6.object({
941
+ taxonId: z6.string().nullable(),
942
+ name: z6.string().nullable(),
943
+ detail: z6.string().nullable(),
944
+ line: z6.number().int().nullable()
581
945
  })
582
946
  )
583
947
  });
948
+ var derivedDiffFieldSchema = z6.enum([
949
+ "canonical_name",
950
+ "scientific_name",
951
+ "authorship",
952
+ "rank",
953
+ "taxonomic_status",
954
+ "accepted_name_usage_id",
955
+ "parent_name_usage_id",
956
+ "family"
957
+ ]);
958
+ var DERIVED_DIFF_FIELDS = derivedDiffFieldSchema.options;
959
+ var diffSampleSchema = z6.object({ taxonId: z6.string(), name: z6.string().nullable() });
960
+ var changedSampleSchema = diffSampleSchema.extend({
961
+ changes: z6.array(
962
+ z6.object({ field: derivedDiffFieldSchema, before: z6.string().nullable(), after: z6.string().nullable() })
963
+ )
964
+ });
965
+ var derivedReportSchema = z6.object({
966
+ rowsRead: z6.number().int().nonnegative(),
967
+ rowsStaged: z6.number().int().nonnegative(),
968
+ rowsRejected: z6.number().int().nonnegative(),
969
+ baseRowCount: z6.number().int().nonnegative(),
970
+ errors: z6.number().int().nonnegative(),
971
+ warnings: z6.number().int().nonnegative(),
972
+ issues: z6.array(derivedIssueSchema),
973
+ diff: z6.object({
974
+ added: z6.object({ count: z6.number().int().nonnegative(), samples: z6.array(diffSampleSchema) }),
975
+ removed: z6.object({ count: z6.number().int().nonnegative(), samples: z6.array(diffSampleSchema) }),
976
+ changed: z6.object({
977
+ count: z6.number().int().nonnegative(),
978
+ byField: z6.record(z6.string(), z6.number().int().nonnegative()),
979
+ samples: z6.array(changedSampleSchema)
980
+ }),
981
+ unchanged: z6.number().int().nonnegative()
982
+ })
983
+ });
984
+ var derivedUploadSchema = z6.object({
985
+ id: z6.string().uuid(),
986
+ backbone: backboneSchema,
987
+ baseVersion: z6.string(),
988
+ slug: z6.string(),
989
+ label: z6.string(),
990
+ notes: z6.string().nullable(),
991
+ /** The snapshot version this upload installs as (`<base>+<slug>`). */
992
+ version: z6.string(),
993
+ originalFilename: z6.string().nullable(),
994
+ sha256: z6.string(),
995
+ sizeBytes: z6.number().int().nonnegative(),
996
+ status: derivedUploadStatusSchema,
997
+ rowsRead: z6.number().int().nonnegative(),
998
+ report: derivedReportSchema.nullable(),
999
+ message: z6.string().nullable(),
1000
+ createdAt: z6.string().datetime(),
1001
+ expiresAt: z6.string().datetime(),
1002
+ importedVersion: z6.string().nullable()
1003
+ });
1004
+ var derivedConfirmBodySchema = z6.object({
1005
+ /** Acknowledge warnings. Never overrides errors. */
1006
+ force: z6.boolean().optional()
1007
+ });
1008
+
1009
+ // ../shared/src/schemas/openrouter.ts
1010
+ import { z as z7 } from "zod";
1011
+ var openRouterModelSchema = z7.object({
1012
+ /** Full model id, e.g. `anthropic/claude-opus-4.8` or the alias
1013
+ * `~anthropic/claude-haiku-latest`. Used verbatim as the OpenRouter model. */
1014
+ id: z7.string(),
1015
+ /** Human label from OpenRouter (falls back to the id). */
1016
+ name: z7.string(),
1017
+ /** Author slug (the part before `/`), with any leading `~` stripped. */
1018
+ author: z7.string(),
1019
+ /** Unix seconds the model was published; used to rank "latest". */
1020
+ created: z7.number(),
1021
+ /** Context window in tokens, when OpenRouter reports it. */
1022
+ contextLength: z7.number().nullable(),
1023
+ /** USD price per PROMPT token (input). Null when OpenRouter doesn't report a
1024
+ * numeric price. 0 = free. */
1025
+ promptPriceUsd: z7.number().nullable(),
1026
+ /** USD price per COMPLETION token (output). */
1027
+ completionPriceUsd: z7.number().nullable(),
1028
+ /** An auto-updating `…-latest` pointer (e.g. `~anthropic/claude-haiku-latest`). */
1029
+ isAlias: z7.boolean()
1030
+ });
584
1031
 
585
1032
  // ../shared/src/normalize.ts
586
1033
  var NULL_SENTINELS = /* @__PURE__ */ new Set(["", "na", "n/a", "null", "-", "\u2014", "unknown", "undet", "undet."]);
@@ -708,21 +1155,21 @@ function detectIdTypeDistribution(samples, opts) {
708
1155
  }
709
1156
 
710
1157
  // ../shared/src/column-map.ts
711
- import { z as z5 } from "zod";
712
- var columnMappingSchema = z5.object({
713
- nameColumn: z5.string().nullable(),
714
- idColumn: z5.string().nullable(),
715
- familyColumn: z5.string().nullable(),
716
- genusColumn: z5.string().nullable(),
717
- rankColumn: z5.string().nullable(),
718
- authorColumn: z5.string().nullable()
719
- });
720
- var detectColumnsBodySchema = z5.object({
721
- headers: z5.array(z5.string().min(1)).min(1).max(200)
722
- });
723
- var detectColumnsResponseSchema = z5.object({
1158
+ import { z as z8 } from "zod";
1159
+ var columnMappingSchema = z8.object({
1160
+ nameColumn: z8.string().nullable(),
1161
+ idColumn: z8.string().nullable(),
1162
+ familyColumn: z8.string().nullable(),
1163
+ genusColumn: z8.string().nullable(),
1164
+ rankColumn: z8.string().nullable(),
1165
+ authorColumn: z8.string().nullable()
1166
+ });
1167
+ var detectColumnsBodySchema = z8.object({
1168
+ headers: z8.array(z8.string().min(1)).min(1).max(200)
1169
+ });
1170
+ var detectColumnsResponseSchema = z8.object({
724
1171
  mapping: columnMappingSchema,
725
- usedLlm: z5.boolean()
1172
+ usedLlm: z8.boolean()
726
1173
  });
727
1174
 
728
1175
  // src/api-client.ts
@@ -775,7 +1222,7 @@ var apiClient = {
775
1222
  const buf = await fs.readFile(filePath);
776
1223
  const fd = new FormData();
777
1224
  const lower = filePath.toLowerCase();
778
- const mime = lower.endsWith(".json") ? "application/json" : "text/csv";
1225
+ const mime = lower.endsWith(".json") ? "application/json" : lower.endsWith(".xlsx") || lower.endsWith(".xls") ? "application/vnd.openxmlformats-officedocument.spreadsheetml.sheet" : "text/csv";
779
1226
  fd.set("file", new Blob([buf], { type: mime }), basename(filePath));
780
1227
  fd.set("config", JSON.stringify(config));
781
1228
  if (name && name.trim()) fd.set("name", name.trim());
@@ -793,6 +1240,7 @@ var apiClient = {
793
1240
  async downloadJob(creds, id, opts) {
794
1241
  const params = new URLSearchParams({ format: opts.format });
795
1242
  if (opts.confirmedOnly) params.set("confirmedOnly", "true");
1243
+ if (opts.dedupe) params.set("dedupe", "true");
796
1244
  if (opts.bundle) params.set("bundle", "true");
797
1245
  if (opts.delimiter) params.set("delimiter", opts.delimiter);
798
1246
  if (opts.columns) params.set("columns", opts.columns);
@@ -1023,8 +1471,9 @@ function trunc(s, n) {
1023
1471
  }
1024
1472
 
1025
1473
  // src/index.ts
1474
+ var { version } = createRequire(import.meta.url)("../package.json");
1026
1475
  var program = new Command();
1027
- program.name("planttaxomatcher").description("PlantTaxoMatcher CLI").version("0.1.1");
1476
+ program.name("planttaxomatcher").description("PlantTaxoMatcher CLI").version(version);
1028
1477
  program.command("login").description("Save a personal token + server URL").option(
1029
1478
  "--token <token>",
1030
1479
  "Personal token (ptm_...). Avoid on shared hosts: it is visible in shell history and the process list. Prefer --token-stdin or the PLANTTAXOMATCHER_TOKEN env var."
@@ -1036,17 +1485,15 @@ program.command("login").description("Save a personal token + server URL").optio
1036
1485
  "--insecure",
1037
1486
  "Allow sending the token over cleartext http to a non-loopback server (NOT recommended).",
1038
1487
  false
1039
- ).action(
1040
- async (opts) => {
1041
- assertServerTransport(opts.server, !!opts.insecure);
1042
- const token = await resolveLoginToken(opts);
1043
- if (!tokenStringSchema.safeParse(token).success) {
1044
- throw new Error("malformed token \u2014 expected ptm_<12 chars>_<32 chars>");
1045
- }
1046
- await writeCredentials({ token, server: opts.server });
1047
- console.log(kleur2.green("\u2713"), `Logged in to ${opts.server}`);
1488
+ ).action(async (opts) => {
1489
+ assertServerTransport(opts.server, !!opts.insecure);
1490
+ const token = await resolveLoginToken(opts);
1491
+ if (!tokenStringSchema.safeParse(token).success) {
1492
+ throw new Error("malformed token \u2014 expected ptm_<12 chars>_<32 chars>");
1048
1493
  }
1049
- );
1494
+ await writeCredentials({ token, server: opts.server });
1495
+ console.log(kleur2.green("\u2713"), `Logged in to ${opts.server}`);
1496
+ });
1050
1497
  program.command("logout").description("Clear saved credentials").action(async () => {
1051
1498
  await clearCredentials();
1052
1499
  console.log(kleur2.green("\u2713"), "Logged out");
@@ -1081,25 +1528,27 @@ program.command("submit <files...>").description(
1081
1528
  "--name <label>",
1082
1529
  "Job name override (single file only; ignored when multiple files match \u2014 the file name is used)"
1083
1530
  ).requiredOption("--name-column <name>", "Column holding the scientific name").option("--id-column <name>", "Column holding a known ID").option("--family-column <name>", "Column holding family").option("--genus-column <name>", "Column holding genus").option("--rank-column <name>", "Column holding rank").option("--author-column <name>", "Column holding authorship").option("--filter-column <name>", "Only process rows where this column matches --filter-value; others are skipped").option("--filter-value <value>", "Value the --filter-column must equal (trimmed, case-insensitive)").option("--author-mode <mode>", "ignore|prefer|strict", "prefer").option("--parallel <n>", "Per-job parallelism", "4").option("--review-mode <mode>", "off|recommended|strict", "recommended").option(
1084
- "--keep-infraspecific",
1085
- "Keep infraspecific accepted taxa (varieties, subspecies, forms) instead of collapsing them up to the species",
1086
- false
1087
- ).option("--allow-llm", "Allow Layer 7 LLM cascade (breaks ties between competing matches)", false).option("--llm-cap-cents <cents>", "Max LLM spend in cents", "500").option("--referential <version>", "WCVP snapshot version to match against (default: latest)").option("--no-watch", "Do not stream progress after submit").option(
1088
- "--dry-run",
1089
- "Preview locally (normalize first rows + detect ID type) and confirm before uploading",
1531
+ "--species-level",
1532
+ "Roll infraspecific accepted taxa (varieties, subspecies, forms) up to their species, so every output row sits at species level",
1090
1533
  false
1091
1534
  ).option(
1092
- "--dry-run-rows <n>",
1093
- "Number of rows to show in the dry-run preview table (default 10)",
1094
- "10"
1095
- ).action(async (files, opts) => {
1535
+ "--ignore-author",
1536
+ "Accept a unique canonical name match whose author couldn't be confirmed instead of sending it to review (an author that CONFLICTS with the matched taxon still reviews)",
1537
+ false
1538
+ ).option("--keep-infraspecific", "Deprecated \u2014 now the default; has no effect", false).option("--allow-llm", "Allow Layer 7 LLM cascade (breaks ties between competing matches)", false).option("--llm-cap-cents <cents>", "Max LLM spend in cents", "500").option(
1539
+ "--referential <version>",
1540
+ "WCVP snapshot version to match against (default: your team's configured default snapshot, else the newest import)"
1541
+ ).option("--no-watch", "Do not stream progress after submit").option("--dry-run", "Preview locally (normalize first rows + detect ID type) and confirm before uploading", false).option("--dry-run-rows <n>", "Number of rows to show in the dry-run preview table (default 10)", "10").action(async (files, opts) => {
1096
1542
  const creds = await requireCredentials();
1097
1543
  const inputs = await expandInputs(files);
1098
1544
  if (inputs.length === 0) throw new Error("no input files");
1099
1545
  if (opts.name && inputs.length > 1) {
1546
+ console.log(kleur2.yellow("!"), "--name ignored for multi-file submit; using each file name as the job name");
1547
+ }
1548
+ if (opts.keepInfraspecific) {
1100
1549
  console.log(
1101
1550
  kleur2.yellow("!"),
1102
- "--name ignored for multi-file submit; using each file name as the job name"
1551
+ "--keep-infraspecific is deprecated and does nothing \u2014 infraspecific taxa are kept by default. Pass --species-level to roll them up to the species."
1103
1552
  );
1104
1553
  }
1105
1554
  if (inputs.length > 1) {
@@ -1145,9 +1594,12 @@ ${file}`));
1145
1594
  allowLlm: !!opts.allowLlm,
1146
1595
  llmCostCapCents: Number(opts.llmCapCents ?? 500),
1147
1596
  reviewMode: String(opts.reviewMode ?? "recommended"),
1148
- // UI/CLI opt-in inverts the config flag: by default we collapse an
1149
- // infraspecific accepted taxon up to its species.
1150
- speciesLevelAcceptedOnly: !opts.keepInfraspecific,
1597
+ // Opt-in: by default an infraspecific accepted taxon is kept at the
1598
+ // rank it resolved to, not rolled up to its species.
1599
+ speciesLevelAcceptedOnly: !!opts.speciesLevel,
1600
+ // Opt-in: an unconfirmed author still routes the match to review
1601
+ // unless the caller declares the author unimportant.
1602
+ acceptUnconfirmedAuthor: !!opts.ignoreAuthor,
1151
1603
  exportConfirmedOnly: false,
1152
1604
  forceReviewFamilies: [],
1153
1605
  ...opts.referential ? { referentialVersion: String(opts.referential) } : {}
@@ -1187,10 +1639,7 @@ program.command("cancel <jobId>").description("Cancel a job. Already-matched row
1187
1639
  const job = await apiClient.cancelJob(creds, jobId);
1188
1640
  console.log(kleur2.green("\u2713"), `cancelled: ${job.id} (status=${job.status})`);
1189
1641
  });
1190
- program.command("download <jobId>").description("Download a job export (CSV, JSON, or NDJSON; optionally bundled with NOTICE.md)").option("--format <fmt>", "csv | xlsx | json | ndjson", "csv").option("--confirmed-only", "only matched rows that are accepted / not pending", false).option("--delimiter <sep>", "CSV separator: comma | semicolon | tab | pipe (csv format only)", "comma").option("--bundle", "wrap in a ZIP with NOTICE.md citing the WCVP snapshot + providers", false).option(
1191
- "--columns <list>",
1192
- "comma-separated result/upload column keys to KEEP (default: all). See --list-columns"
1193
- ).option(
1642
+ program.command("download <jobId>").description("Download a job export (CSV, JSON, or NDJSON; optionally bundled with NOTICE.md)").option("--format <fmt>", "csv | xlsx | json | ndjson", "csv").option("--confirmed-only", "only matched rows that are accepted / not pending", false).option("--dedupe", "collapse rows that resolved to the same accepted taxon to one line", false).option("--delimiter <sep>", "CSV separator: comma | semicolon | tab | pipe (csv format only)", "comma").option("--bundle", "wrap in a ZIP with NOTICE.md citing the WCVP snapshot + providers", false).option("--columns <list>", "comma-separated result/upload column keys to KEEP (default: all). See --list-columns").option(
1194
1643
  "--wcvp-extra <list>",
1195
1644
  "comma-separated extra WCVP fields to append as wcvp_<key> columns (e.g. ipni_id,powo_id). See --list-columns"
1196
1645
  ).option("--list-columns", "print the columns available for this job and exit", false).option("--output <path>", "write to this path; default is the server-provided filename in CWD").action(async (jobId, opts) => {
@@ -1212,6 +1661,7 @@ program.command("download <jobId>").description("Download a job export (CSV, JSO
1212
1661
  throw new Error(`--format must be csv, xlsx, json or ndjson (got ${String(format)})`);
1213
1662
  }
1214
1663
  const confirmedOnly = !!opts.confirmedOnly;
1664
+ const dedupe = !!opts.dedupe;
1215
1665
  const bundle = !!opts.bundle;
1216
1666
  const delimiter = String(opts.delimiter ?? "comma");
1217
1667
  if (!["comma", "semicolon", "tab", "pipe"].includes(delimiter)) {
@@ -1220,6 +1670,7 @@ program.command("download <jobId>").description("Download a job export (CSV, JSO
1220
1670
  const { filename, body } = await apiClient.downloadJob(creds, jobId, {
1221
1671
  format,
1222
1672
  confirmedOnly,
1673
+ dedupe,
1223
1674
  bundle,
1224
1675
  ...format === "csv" && delimiter !== "comma" ? { delimiter } : {},
1225
1676
  ...opts.columns ? { columns: String(opts.columns) } : {},
@@ -1269,9 +1720,7 @@ async function resolveLoginToken(opts) {
1269
1720
  const env = process.env.PLANTTAXOMATCHER_TOKEN;
1270
1721
  if (env && env.trim()) return env.trim();
1271
1722
  if (!process.stdin.isTTY) {
1272
- throw new Error(
1273
- "no token provided. Pass --token-stdin, set PLANTTAXOMATCHER_TOKEN, or run interactively."
1274
- );
1723
+ throw new Error("no token provided. Pass --token-stdin, set PLANTTAXOMATCHER_TOKEN, or run interactively.");
1275
1724
  }
1276
1725
  const entered = await password({ message: "Personal token (ptm_...)", mask: true });
1277
1726
  return entered.trim();
@@ -1298,6 +1747,7 @@ async function expandInputs(patterns) {
1298
1747
  }
1299
1748
  async function streamJob(creds, jobId) {
1300
1749
  console.log(kleur2.cyan("\u2192"), `Streaming progress for ${jobId} \u2026`);
1750
+ let lastLineWidth = 0;
1301
1751
  for await (const evt of apiClient.streamJob(creds, jobId)) {
1302
1752
  const t = String(evt.type ?? "");
1303
1753
  if (t === "heartbeat") continue;
@@ -1315,7 +1765,10 @@ async function streamJob(creds, jobId) {
1315
1765
  } else if (t === "progress") {
1316
1766
  const p = evt.processedQueries ?? evt.processedRows ?? 0;
1317
1767
  const total = evt.totalQueries ?? evt.totalRows ?? 0;
1318
- process.stdout.write(`\r progress: ${p}/${total} `);
1768
+ const phase = p === 0 && typeof evt.phase === "string" ? ` (${evt.phase}\u2026)` : "";
1769
+ const line = ` progress: ${p}/${total}${phase}`;
1770
+ process.stdout.write(`\r${line.padEnd(lastLineWidth)}`);
1771
+ lastLineWidth = line.length;
1319
1772
  } else if (t === "completed") {
1320
1773
  process.stdout.write("\n");
1321
1774
  console.log(kleur2.green("\u2713"), "completed");