@plantnet/planttaxomatcher 0.3.0 → 0.4.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js CHANGED
@@ -251,6 +251,8 @@ var wcvpSnapshotSummarySchema = z4.object({
251
251
  importedAt: z4.string().datetime(),
252
252
  /** Newest OFFICIAL snapshot (derived ones never count as latest). */
253
253
  isLatest: z4.boolean(),
254
+ /** The version a job uses when none is named: the team's default if still visible, else the latest. */
255
+ isDefault: z4.boolean(),
254
256
  kind: z4.enum(["official", "derived"]),
255
257
  baseVersion: z4.string().nullable(),
256
258
  label: z4.string().nullable(),
@@ -1046,6 +1048,8 @@ var wfoSnapshotSummarySchema = z8.object({
1046
1048
  recordCount: z8.number().int().nonnegative(),
1047
1049
  importedAt: z8.string().datetime(),
1048
1050
  isLatest: z8.boolean(),
1051
+ /** The version a job uses when none is named: the team's default if still visible, else the latest. */
1052
+ isDefault: z8.boolean(),
1049
1053
  kind: snapshotKindSchema,
1050
1054
  baseVersion: z8.string().nullable(),
1051
1055
  label: z8.string().nullable(),
@@ -1358,6 +1362,87 @@ function parseEnvFlag(raw, fallback) {
1358
1362
  // ../shared/src/public-server.ts
1359
1363
  var DEFAULT_SERVER_URL = "https://planttaxomatcher.plantnet.org";
1360
1364
 
1365
+ // ../shared/src/export-columns.ts
1366
+ import { z as z11 } from "zod";
1367
+ var RESULT_COLUMN_DESCRIPTIONS = {
1368
+ planttaxomatcher_job_id: "Id of the job this row belongs to.",
1369
+ planttaxomatcher_input_normalized: "The input name as it was matched, after cleaning (spacing, case, qualifiers removed).",
1370
+ planttaxomatcher_input_qualifier: "Identification qualifier found in the input: cf, aff, sensu_lato, sensu_stricto, aggregate, complex, sp, spp, indet, cultivar, hybrid_formula, informal. Empty when none.",
1371
+ planttaxomatcher_parse_quality: "How cleanly the name parsed (gnparser): 1 clean, 2 minor issues, 3 significant issues, 4 badly formed; empty when unparsed.",
1372
+ planttaxomatcher_match_status: "matched, ambiguous (several candidates, none chosen), no_match, error, or skipped (filtered out, or no usable name).",
1373
+ planttaxomatcher_evidence_type: "How the match was found: local_exact, local_canonical_unique, local_fuzzy, external_validated, external_fuzzy, team_history, cross_backbone, llm.",
1374
+ planttaxomatcher_grade: "A (confident, accepted automatically), B (plausible, needs review) or C (weakest: a language-model suggestion or genus only).",
1375
+ planttaxomatcher_review_status: "not_required (grade A), pending (waiting for review), accepted, rejected, or overridden (a reviewer picked another taxon).",
1376
+ planttaxomatcher_confidence: "Match score between 0 and 1.",
1377
+ planttaxomatcher_score_breakdown: "JSON of the score components: canonical, author, rank and family similarity, provider agreement.",
1378
+ planttaxomatcher_layer: "The cascade layer family that settled the row (same values as the evidence type).",
1379
+ planttaxomatcher_reason: "One-line explanation of the decision.",
1380
+ planttaxomatcher_flags: "Comma-separated reasons a match needs a look, e.g. author-unconfirmed, author-mismatch, genus-fallback.",
1381
+ planttaxomatcher_matched_identifier_namespace: "Namespace of the matched name\u2019s identifier, e.g. wcvp:taxonID or wfo:taxonID.",
1382
+ planttaxomatcher_matched_identifier: "Identifier of the matched name (possibly a synonym).",
1383
+ planttaxomatcher_accepted_identifier_namespace: "Namespace of the accepted taxon\u2019s identifier: wcvp:acceptedNameUsageID or wfo:acceptedNameUsageID.",
1384
+ planttaxomatcher_accepted_identifier: "Identifier of the accepted taxon, in the job\u2019s backbone (WCVP or WFO).",
1385
+ plantnet_identifiable: "true when Pl@ntNet can identify the accepted taxon from photos (when Pl@ntNet is enabled).",
1386
+ plantnet_id: "Pl@ntNet species id of the accepted taxon (when Pl@ntNet is enabled).",
1387
+ wcvp_matched_name: "The backbone name the input matched, which may be a synonym.",
1388
+ wcvp_matched_taxonomic_status: "Status of the matched name in the backbone, e.g. Accepted, Synonym or Illegitimate.",
1389
+ wcvp_accepted_name: "The accepted name the match resolves to, without author.",
1390
+ wcvp_accepted_author: "Author of the accepted name.",
1391
+ wcvp_accepted_name_with_author: "The accepted name and its author in one cell, ready to cite, e.g. Calicotome spinosa (L.) Link. The column to hand back as the cleaned name.",
1392
+ wcvp_accepted_taxon_id: "WCVP id (plant_name_id) of the accepted taxon. Empty for WFO jobs, whose id is in planttaxomatcher_accepted_identifier.",
1393
+ wcvp_accepted_usage_id: "Same id as wcvp_accepted_taxon_id, under its Darwin Core name (acceptedNameUsageID).",
1394
+ ipni_lsid: "IPNI LSID of the accepted name: urn:lsid:ipni.org:names: followed by its IPNI id.",
1395
+ powo_url: "Plants of the World Online page of the accepted taxon.",
1396
+ wcvp_family: "Family of the accepted taxon.",
1397
+ wcvp_rank: "Rank of the accepted taxon, e.g. Species, Subspecies, Variety or Genus.",
1398
+ planttaxomatcher_alternatives: "JSON list of the other candidates, mostly for ambiguous rows.",
1399
+ planttaxomatcher_reviewed: "true when a person reviewed the row.",
1400
+ planttaxomatcher_referential_version: "Backbone version the job matched against, e.g. wcvp-v14.",
1401
+ planttaxomatcher_normalizer_version: "Version of the name normaliser, for reproducibility.",
1402
+ planttaxomatcher_parser_version: "Version of the name parser, for reproducibility.",
1403
+ planttaxomatcher_matcher_version: "Version of the matching cascade, for reproducibility.",
1404
+ planttaxomatcher_scoring_version: "Version of the grading rules, for reproducibility."
1405
+ };
1406
+ var taxonLevelSchema = z11.enum(["all", "species_below", "species_only"]);
1407
+ var WCVP_EXTRA_DESCRIPTIONS = {
1408
+ ipni_id: "IPNI id of the name (the part after urn:lsid:ipni.org:names:).",
1409
+ powo_id: "Plants of the World Online id of the taxon.",
1410
+ plant_name_id: "WCVP id of the record.",
1411
+ accepted_plant_name_id: "WCVP id of the accepted name the record points to.",
1412
+ basionym_plant_name_id: "WCVP id of the basionym (the original name this one is based on).",
1413
+ parent_plant_name_id: "WCVP id of the parent taxon, e.g. the species of a variety.",
1414
+ taxon_authors: "Full author string of the name.",
1415
+ primary_author: "Author or authors who published the name.",
1416
+ parenthetical_author: "Author of the basionym, shown in parentheses in the author string.",
1417
+ publication_author: "Author of the book or article where the name was published, when different.",
1418
+ first_published: "Year of first publication.",
1419
+ place_of_publication: "Journal or book where the name was published.",
1420
+ volume_and_page: "Volume and page of the publication.",
1421
+ geographic_area: "Native distribution, as a short narrative statement.",
1422
+ climate_description: "Habitat or climate type, from published habitat information.",
1423
+ lifeform_description: "Life form, in a modified Raunki\xE6r system (e.g. phanerophyte).",
1424
+ homotypic_synonym: "Whether the name is a homotypic synonym (same type as the accepted name).",
1425
+ replaced_synonym_author: "Author of the replaced synonym, for a replacement name.",
1426
+ nomenclatural_remarks: "Nomenclatural notes, e.g. nom. illeg.",
1427
+ genus_hybrid: "Hybrid marker (\xD7) of the genus, when it is a hybrid.",
1428
+ species_hybrid: "Hybrid marker (\xD7) of the species, when it is a hybrid.",
1429
+ hybrid_formula: "Hybrid formula, e.g. Mentha aquatica \xD7 M. spicata.",
1430
+ infraspecies: "Infraspecific epithet.",
1431
+ infraspecific_rank: "Infraspecific rank, e.g. subsp., var. or f.",
1432
+ species: "Species epithet.",
1433
+ taxon_name: "Full name without author.",
1434
+ reviewed: "Whether Kew has peer-reviewed the family of this taxon.",
1435
+ pn_change: "Team-edited versions only: what the team changed on this record (codes such as re_accept;author).",
1436
+ pn_source: "Team-edited versions only: source of the change.",
1437
+ pn_taxon_id: "Team-edited versions only: the team\u2019s own taxon id.",
1438
+ wcvp13_taxon_status: "Team-edited versions only: taxonomic status before the edit (WCVP v13).",
1439
+ wcvp13_accepted_plant_name_id: "Team-edited versions only: accepted name id before the edit (WCVP v13).",
1440
+ wcvp13_taxon_name: "Team-edited versions only: name before the edit (WCVP v13).",
1441
+ wcvp13_taxon_authors: "Team-edited versions only: authors before the edit (WCVP v13).",
1442
+ wcvp13_family: "Team-edited versions only: family before the edit (WCVP v13).",
1443
+ wcvp13_genus: "Team-edited versions only: genus before the edit (WCVP v13)."
1444
+ };
1445
+
1361
1446
  // src/errors.ts
1362
1447
  import { CommanderError } from "commander";
1363
1448
  var EXIT = {
@@ -1463,10 +1548,15 @@ function downloadFilename(contentDisposition, jobId, opts) {
1463
1548
  if (suggested && suggested !== "." && suggested !== "..") return suggested;
1464
1549
  return `planttaxomatcher_${jobId}.${opts.bundle ? "zip" : opts.format}`;
1465
1550
  }
1551
+ function filterParams(filters, params = new URLSearchParams()) {
1552
+ if (filters.confirmedOnly) params.set("confirmedOnly", "true");
1553
+ if (filters.resolvedOnly) params.set("notEmptyOnly", "true");
1554
+ if (filters.dedupe) params.set("dedupe", "true");
1555
+ if (filters.taxonLevel && filters.taxonLevel !== "all") params.set("taxonLevel", filters.taxonLevel);
1556
+ return params;
1557
+ }
1466
1558
  function downloadQuery(opts) {
1467
- const params = new URLSearchParams({ format: opts.format });
1468
- if (opts.confirmedOnly) params.set("confirmedOnly", "true");
1469
- if (opts.dedupe) params.set("dedupe", "true");
1559
+ const params = filterParams(opts, new URLSearchParams({ format: opts.format }));
1470
1560
  if (opts.bundle) params.set("bundle", "true");
1471
1561
  if (opts.delimiter) params.set("delimiter", opts.delimiter);
1472
1562
  if (opts.columns) params.set("columns", opts.columns);
@@ -1567,6 +1657,15 @@ function createApiClient(options = {}) {
1567
1657
  },
1568
1658
  getJob: (c, id) => call(c, `/v1/jobs/${encodeURIComponent(id)}`),
1569
1659
  downloadColumns: (c, id) => call(c, `/v1/jobs/${encodeURIComponent(id)}/download/columns`),
1660
+ downloadCounts(c, id, filters) {
1661
+ const query = filterParams(filters);
1662
+ const suffix = query.size > 0 ? `?${query}` : "";
1663
+ return call(c, `/v1/jobs/${encodeURIComponent(id)}/download/count${suffix}`);
1664
+ },
1665
+ listVersions: (c, backbone) => call(c, `/v1/${backbone}/snapshots`),
1666
+ async listAreas(c) {
1667
+ return (await call(c, "/v1/wgsrpd/areas")).areas;
1668
+ },
1570
1669
  pauseJob: (c, id) => call(c, `/v1/jobs/${encodeURIComponent(id)}/pause`, { method: "POST" }),
1571
1670
  resumeJob: (c, id) => call(c, `/v1/jobs/${encodeURIComponent(id)}/resume`, { method: "POST" }),
1572
1671
  cancelJob: (c, id) => call(c, `/v1/jobs/${encodeURIComponent(id)}/cancel`, { method: "POST" }),
@@ -1805,7 +1904,7 @@ function defaultDeps() {
1805
1904
 
1806
1905
  // src/program.ts
1807
1906
  import { Command } from "commander";
1808
- import kleur9 from "kleur";
1907
+ import kleur10 from "kleur";
1809
1908
 
1810
1909
  // src/output.ts
1811
1910
  import kleur from "kleur";
@@ -1980,11 +2079,124 @@ function registerAuthCommands(program, deps2) {
1980
2079
  });
1981
2080
  }
1982
2081
 
2082
+ // src/commands/catalog.ts
2083
+ import kleur3 from "kleur";
2084
+
2085
+ // src/catalog.ts
2086
+ function parseBackbone(value) {
2087
+ const parsed = backboneSchema.safeParse(value.trim().toLowerCase());
2088
+ if (!parsed.success) throw new CliError(`--backbone must be wcvp or wfo (got "${value}")`, EXIT.usage);
2089
+ return parsed.data;
2090
+ }
2091
+ function toVersionEntries(backbone, summaries) {
2092
+ return summaries.map((s) => ({
2093
+ backbone,
2094
+ version: s.version,
2095
+ kind: s.kind,
2096
+ label: s.label,
2097
+ baseVersion: s.baseVersion,
2098
+ isLatest: s.isLatest,
2099
+ // Servers before isDefault fall back to "latest", which is what they used.
2100
+ isDefault: s.isDefault ?? s.isLatest,
2101
+ recordCount: s.recordCount,
2102
+ importedAt: s.importedAt
2103
+ }));
2104
+ }
2105
+ var MAX_LISTED = 12;
2106
+ function requireVersion(entries, backbone, requested) {
2107
+ const found = entries.find((e) => e.version === requested);
2108
+ if (found) return found;
2109
+ const available = entries.map((e) => e.version);
2110
+ const listed = available.slice(0, MAX_LISTED).join(", ") + (available.length > MAX_LISTED ? ", \u2026" : "");
2111
+ throw new CliError(
2112
+ `no ${backbone.toUpperCase()} version "${requested}"${available.length ? ` (available: ${listed})` : ""}`,
2113
+ EXIT.usage,
2114
+ `Run \`planttaxomatcher versions --backbone ${backbone}\` to see them all.`
2115
+ );
2116
+ }
2117
+ function matches(area, needle) {
2118
+ return [area.code, area.name, area.region, area.continent].some((field) => field?.toLowerCase().includes(needle));
2119
+ }
2120
+ function searchAreas(areas, search) {
2121
+ const needle = search?.trim().toLowerCase();
2122
+ return needle ? areas.filter((area) => matches(area, needle)) : areas;
2123
+ }
2124
+ function requireArea(areas, requested) {
2125
+ const code = requested.trim().toUpperCase();
2126
+ const found = areas.find((a) => a.code.toUpperCase() === code);
2127
+ if (found) return found;
2128
+ const suggestions = searchAreas(areas, requested).slice(0, 5).map((a) => `${a.code} (${a.name ?? "?"})`);
2129
+ throw new CliError(
2130
+ `unknown area "${requested}"${suggestions.length ? `; did you mean ${suggestions.join(", ")}?` : ""}`,
2131
+ EXIT.usage,
2132
+ `--area takes a WGSRPD level-3 code: \`planttaxomatcher areas --search ${requested.trim()}\`.`
2133
+ );
2134
+ }
2135
+ function describeVersion(e) {
2136
+ const tags = [e.isDefault ? "default" : null, e.isLatest ? "latest" : null].filter(Boolean).join(", ");
2137
+ const origin = e.kind === "derived" ? `team-edited from ${e.baseVersion ?? "?"}` : "official";
2138
+ const label = e.label ? ` \u201C${e.label}\u201D` : "";
2139
+ return ` ${e.version.padEnd(28)} ${origin}${label}${tags ? ` [${tags}]` : ""} ${e.recordCount.toLocaleString("en")} names, imported ${e.importedAt.slice(0, 10)}`;
2140
+ }
2141
+ function versionLines(entries) {
2142
+ const lines = [];
2143
+ for (const backbone of ["wcvp", "wfo"]) {
2144
+ const ofBackbone = entries.filter((e) => e.backbone === backbone);
2145
+ if (ofBackbone.length === 0) continue;
2146
+ lines.push(backbone.toUpperCase(), ...ofBackbone.map(describeVersion));
2147
+ }
2148
+ return lines;
2149
+ }
2150
+ function areaLine(area) {
2151
+ const place2 = [area.region, area.continent].filter(Boolean).join(", ");
2152
+ return `${area.code.padEnd(4)} ${area.name ?? ""}${place2 ? ` (${place2})` : ""}`;
2153
+ }
2154
+
2155
+ // src/commands/catalog.ts
2156
+ async function fetchVersions(deps2, creds, backbone) {
2157
+ return toVersionEntries(backbone, await deps2.api.listVersions(creds, backbone));
2158
+ }
2159
+ function registerCatalogCommands(program, deps2) {
2160
+ program.command("versions").description(
2161
+ "List the backbone versions a job can match against (--referential), marking the one used by default"
2162
+ ).option("--backbone <name>", "Only this backbone: wcvp or wfo (default: both)").option("--json", "Print the versions as a JSON array", false).action(async (opts) => {
2163
+ const out = createOutput(deps2.stdout, deps2.stderr, opts.json, deps2.now);
2164
+ const backbones = opts.backbone ? [parseBackbone(opts.backbone)] : backboneSchema.options;
2165
+ const creds = await requireCredentials(deps2.store, deps2.env);
2166
+ const entries = (await Promise.all(backbones.map((b) => fetchVersions(deps2, creds, b)))).flat();
2167
+ if (out.json) return out.data(entries);
2168
+ if (entries.length === 0) return out.info(kleur3.gray("No backbone version is installed on this server."));
2169
+ for (const line of versionLines(entries)) out.info(line);
2170
+ });
2171
+ program.command("areas").description("List the WGSRPD level-3 areas (botanical countries) that --area accepts").option("--search <text>", "Only areas whose code, name, region or continent contains this text").option("--json", "Print the areas as a JSON array", false).action(async (opts) => {
2172
+ const out = createOutput(deps2.stdout, deps2.stderr, opts.json, deps2.now);
2173
+ const creds = await requireCredentials(deps2.store, deps2.env);
2174
+ const areas = searchAreas(await deps2.api.listAreas(creds), opts.search);
2175
+ if (out.json) return out.data(areas);
2176
+ if (areas.length === 0) return out.info(kleur3.gray("No matching area."));
2177
+ for (const area of areas) out.info(areaLine(area));
2178
+ });
2179
+ }
2180
+
1983
2181
  // src/commands/download.ts
1984
2182
  import { writeFile } from "fs/promises";
1985
- import kleur3 from "kleur";
2183
+ import kleur4 from "kleur";
1986
2184
  var FORMATS = ["csv", "xlsx", "json", "ndjson"];
1987
2185
  var DELIMITERS = ["comma", "semicolon", "tab", "pipe"];
2186
+ var TAXON_LEVELS = taxonLevelSchema.options;
2187
+ function toDownloadFilters(opts) {
2188
+ const parsed = taxonLevelSchema.safeParse(opts.taxonLevel);
2189
+ if (!parsed.success) {
2190
+ throw new CliError(`--taxon-level must be ${TAXON_LEVELS.join(", ")} (got ${opts.taxonLevel})`, EXIT.usage);
2191
+ }
2192
+ const taxonLevel = parsed.data;
2193
+ return {
2194
+ confirmedOnly: opts.confirmedOnly,
2195
+ resolvedOnly: opts.resolvedOnly,
2196
+ dedupe: opts.dedupe,
2197
+ ...taxonLevel !== "all" ? { taxonLevel } : {}
2198
+ };
2199
+ }
1988
2200
  function toDownloadOptions(opts) {
1989
2201
  const format = opts.format;
1990
2202
  if (!FORMATS.includes(format)) {
@@ -1996,43 +2208,90 @@ function toDownloadOptions(opts) {
1996
2208
  }
1997
2209
  return {
1998
2210
  format,
1999
- confirmedOnly: opts.confirmedOnly,
2000
- dedupe: opts.dedupe,
2211
+ ...toDownloadFilters(opts),
2001
2212
  bundle: opts.bundle,
2002
2213
  ...format === "csv" && delimiter !== "comma" ? { delimiter } : {},
2003
2214
  ...opts.columns ? { columns: opts.columns } : {},
2004
2215
  ...opts.wcvpExtra ? { wcvpExtra: opts.wcvpExtra } : {}
2005
2216
  };
2006
2217
  }
2218
+ function describeColumns(columns) {
2219
+ return {
2220
+ result: columns.result.map((c) => ({
2221
+ ...c,
2222
+ description: c.description ?? RESULT_COLUMN_DESCRIPTIONS[c.key] ?? null
2223
+ })),
2224
+ original: columns.original,
2225
+ wcvpExtra: columns.wcvpExtra.map((f) => ({
2226
+ ...f,
2227
+ description: f.description ?? WCVP_EXTRA_DESCRIPTIONS[f.key] ?? null
2228
+ }))
2229
+ };
2230
+ }
2007
2231
  function printColumns(out, columns) {
2008
- out.info(kleur3.bold("Result columns (--columns):"));
2009
- for (const column of columns.result) out.info(` ${column.key} ${kleur3.gray(`(${column.group})`)}`);
2232
+ const line = (key, description) => out.info(` ${key}${description ? `
2233
+ ${kleur4.gray(description)}` : ""}`);
2234
+ out.info(kleur4.bold("Result columns (--columns):"));
2235
+ for (const column of columns.result) line(column.key, column.description);
2010
2236
  if (columns.original.length > 0) {
2011
- out.info(kleur3.bold("\nYour upload columns (--columns):"));
2012
- for (const key of columns.original) out.info(` ${key}`);
2237
+ out.info(kleur4.bold("\nYour upload columns (--columns):"));
2238
+ for (const key of columns.original) line(key);
2239
+ }
2240
+ out.info(kleur4.bold("\nWCVP fields to append (--wcvp-extra), exported as wcvp_<field>:"));
2241
+ for (const field of columns.wcvpExtra) line(field.key, field.description);
2242
+ }
2243
+ function printCounts(out, counts) {
2244
+ const rows = [
2245
+ ["all rows", counts.total, ""],
2246
+ ["confirmed only", counts.confirmed, "--confirmed-only"],
2247
+ ["resolved only", counts.resolved, "--resolved-only"],
2248
+ ["one row per accepted taxon", counts.deduped, "--dedupe"],
2249
+ ["with the filters given", counts.selected, ""]
2250
+ ];
2251
+ for (const [label, count, flag] of rows) {
2252
+ out.info(` ${label.padEnd(28)} ${String(count).padStart(8)} ${kleur4.gray(flag)}`);
2013
2253
  }
2014
- out.info(kleur3.bold("\nWCVP extra fields (--wcvp-extra):"));
2015
- for (const field of columns.wcvpExtra) out.info(` ${field.key} ${kleur3.gray(`(${field.group})`)}`);
2016
2254
  }
2017
2255
  function registerDownloadCommand(program, deps2) {
2018
- program.command("download <jobId>").description("Download a job export (CSV, XLSX, JSON or NDJSON; optionally zipped with NOTICE.md)").option("--format <fmt>", FORMATS.join(" | "), "csv").option("--confirmed-only", "Only matched rows that are accepted / not pending", false).option("--dedupe", "Collapse rows that resolved to the same accepted taxon to one line", false).option("--delimiter <sep>", `CSV separator: ${DELIMITERS.join(" | ")} (csv only)`, "comma").option("--bundle", "Wrap in a ZIP with NOTICE.md citing the WCVP snapshot and providers", false).option(
2256
+ program.command("download <jobId>").description("Download a job export (CSV, XLSX, JSON or NDJSON; optionally zipped with NOTICE.md)").option("--format <fmt>", FORMATS.join(" | "), "csv").option(
2257
+ "--confirmed-only",
2258
+ "Confirmed rows only: accepted automatically (grade A) or by a reviewer, or overridden",
2259
+ false
2260
+ ).option(
2261
+ "--resolved-only",
2262
+ "Resolved rows only: drop rows with no accepted name (no match, error, rejected)",
2263
+ false
2264
+ ).option(
2265
+ "--dedupe",
2266
+ "Remove duplicate accepted taxa: rows resolving to the same accepted taxon (synonyms, subspecies\u2026) become one line; unmatched rows are kept",
2267
+ false
2268
+ ).option(
2269
+ "--taxon-level <level>",
2270
+ "Keep rows whose accepted taxon is at this level: all, species_below (species and infraspecific, drops genus/family), species_only. Unmatched rows are kept; add --resolved-only to drop them",
2271
+ "all"
2272
+ ).option("--delimiter <sep>", `CSV separator: ${DELIMITERS.join(" | ")} (csv only)`, "comma").option("--bundle", "Wrap in a ZIP with NOTICE.md citing the WCVP snapshot and providers", false).option(
2019
2273
  "--columns <list>",
2020
2274
  "Comma-separated result/upload column keys to keep (default: all). See --list-columns"
2021
2275
  ).option(
2022
2276
  "--wcvp-extra <list>",
2023
2277
  "Comma-separated extra WCVP fields appended as wcvp_<key> columns (e.g. ipni_id,powo_id). See --list-columns"
2024
- ).option("--list-columns", "Print the columns available for this job and exit", false).option(
2278
+ ).option("--list-columns", "Print the columns available for this job, with what each holds, and exit", false).option("--counts", "Print how many rows each filter keeps (as the web download dialog shows) and exit", false).option(
2025
2279
  "--output <path>",
2026
2280
  "Write to this path (default: the server-provided file name in the current directory)"
2027
- ).option("--json", "Print the result (or --list-columns) as JSON", false).action(async (jobId, opts) => {
2281
+ ).option("--json", "Print the result (or --list-columns / --counts) as JSON", false).action(async (jobId, opts) => {
2028
2282
  const out = createOutput(deps2.stdout, deps2.stderr, opts.json, deps2.now);
2029
2283
  const download = toDownloadOptions(opts);
2030
2284
  const creds = await requireCredentials(deps2.store, deps2.env);
2031
2285
  if (opts.listColumns) {
2032
- const columns = await deps2.api.downloadColumns(creds, jobId);
2286
+ const columns = describeColumns(await deps2.api.downloadColumns(creds, jobId));
2033
2287
  if (out.json) return out.data(columns);
2034
2288
  return printColumns(out, columns);
2035
2289
  }
2290
+ if (opts.counts) {
2291
+ const counts = await deps2.api.downloadCounts(creds, jobId, toDownloadFilters(opts));
2292
+ if (out.json) return out.data(counts);
2293
+ return printCounts(out, counts);
2294
+ }
2036
2295
  const { filename, body } = await deps2.api.downloadJob(creds, jobId, download);
2037
2296
  const path = opts.output ?? filename;
2038
2297
  await writeFile(path, body);
@@ -2042,7 +2301,7 @@ function registerDownloadCommand(program, deps2) {
2042
2301
  }
2043
2302
 
2044
2303
  // src/commands/jobs.ts
2045
- import kleur5 from "kleur";
2304
+ import kleur6 from "kleur";
2046
2305
 
2047
2306
  // src/flags.ts
2048
2307
  function parseIntegerFlag(value, flag, range = {}) {
@@ -2056,7 +2315,7 @@ function parseIntegerFlag(value, flag, range = {}) {
2056
2315
  }
2057
2316
 
2058
2317
  // src/watch.ts
2059
- import kleur4 from "kleur";
2318
+ import kleur5 from "kleur";
2060
2319
  var STOP = /* @__PURE__ */ new Set(["completed", "failed", "cancelled", "paused"]);
2061
2320
  function isStop(status) {
2062
2321
  return typeof status === "string" && STOP.has(status);
@@ -2079,15 +2338,15 @@ function render(out, frame) {
2079
2338
  }
2080
2339
  if (frame.type === "status") {
2081
2340
  out.progress.done();
2082
- out.info(`${kleur4.gray("\u2022")} ${String(frame.status)}`);
2341
+ out.info(`${kleur5.gray("\u2022")} ${String(frame.status)}`);
2083
2342
  } else if (frame.type === "error") {
2084
2343
  out.progress.done();
2085
- out.info(`${kleur4.red("error:")} ${String(frame.message ?? "job failed")}`);
2344
+ out.info(`${kleur5.red("error:")} ${String(frame.message ?? "job failed")}`);
2086
2345
  }
2087
2346
  }
2088
2347
  async function watchJob(api, out, creds, jobId) {
2089
2348
  out.progress.done();
2090
- if (!out.json) out.info(`${kleur4.cyan("\u2192")} Streaming progress for ${jobId} \u2026`);
2349
+ if (!out.json) out.info(`${kleur5.cyan("\u2192")} Streaming progress for ${jobId} \u2026`);
2091
2350
  for await (const frame of api.streamJob(creds, jobId)) {
2092
2351
  if (frame.type === "heartbeat") continue;
2093
2352
  const status = stopStatusOf(frame);
@@ -2108,8 +2367,8 @@ function finish(out, status) {
2108
2367
  out.progress.done();
2109
2368
  if (!out.json) {
2110
2369
  if (status === "completed") out.success("completed");
2111
- else if (status === "paused") out.info(kleur4.yellow("paused \u2014 resume it, then watch again"));
2112
- else out.info(kleur4.red(`\u2717 ${status}`));
2370
+ else if (status === "paused") out.info(kleur5.yellow("paused \u2014 resume it, then watch again"));
2371
+ else out.info(kleur5.red(`\u2717 ${status}`));
2113
2372
  }
2114
2373
  return status;
2115
2374
  }
@@ -2131,7 +2390,7 @@ function watchOutcomeError(results) {
2131
2390
  }
2132
2391
  function listLine(job) {
2133
2392
  const matched = `${String(job.matchedRows).padStart(6)}/${String(job.totalRows).padStart(6)} matched`;
2134
- return `${job.id} ${job.status.padEnd(10)} ${matched} ${kleur5.gray(job.name ?? "")}`;
2393
+ return `${job.id} ${job.status.padEnd(10)} ${matched} ${kleur6.gray(job.name ?? "")}`;
2135
2394
  }
2136
2395
  var JOB_ACTIONS = {
2137
2396
  pause: { description: "Pause a running job (the worker stops between match queries)", past: "paused" },
@@ -2154,7 +2413,7 @@ function registerJobCommands(program, deps2) {
2154
2413
  const creds = await requireCredentials(deps2.store, deps2.env);
2155
2414
  const jobs = await deps2.api.listJobs(creds, { limit, ...status ? { status: status.data } : {} });
2156
2415
  if (out.json) return out.data(jobs);
2157
- if (jobs.length === 0) return out.info(kleur5.gray("No jobs."));
2416
+ if (jobs.length === 0) return out.info(kleur6.gray("No jobs."));
2158
2417
  for (const job of jobs) out.info(listLine(job));
2159
2418
  });
2160
2419
  program.command("status <jobId>").description("Print a job snapshot as JSON (counts, status, timestamps)").option("--json", "Accepted for symmetry: the output is always JSON", false).action(async (jobId) => {
@@ -2182,7 +2441,7 @@ function registerJobCommands(program, deps2) {
2182
2441
  }
2183
2442
 
2184
2443
  // src/commands/skills.ts
2185
- import kleur6 from "kleur";
2444
+ import kleur7 from "kleur";
2186
2445
 
2187
2446
  // src/skills/manage.ts
2188
2447
  import { promises as fs3 } from "fs";
@@ -2201,11 +2460,31 @@ async function walk(dir, root = dir) {
2201
2460
  }
2202
2461
  return found.sort();
2203
2462
  }
2463
+ function table(rows, keyOf) {
2464
+ const lines = Object.entries(rows).map(([key, description]) => `| \`${keyOf(key)}\` | ${description} |`);
2465
+ return ["| Column | What it holds |", "| --- | --- |", ...lines].join("\n");
2466
+ }
2467
+ function exportColumnsMarkdown() {
2468
+ return [
2469
+ "## Every column",
2470
+ "",
2471
+ table(RESULT_COLUMN_DESCRIPTIONS, (key) => key),
2472
+ "",
2473
+ "## Fields `--wcvp-extra` can append",
2474
+ "",
2475
+ "Columns of the accepted taxon's WCVP record, exported as `wcvp_<field>`, e.g. `--wcvp-extra ipni_id,powo_id`.",
2476
+ "",
2477
+ table(WCVP_EXTRA_DESCRIPTIONS, (key) => key)
2478
+ ].join("\n");
2479
+ }
2204
2480
  async function renderSkill(sourceDir, version2) {
2205
2481
  const files = /* @__PURE__ */ new Map();
2206
2482
  for (const path of await walk(sourceDir)) {
2207
2483
  const text = await fs2.readFile(join3(sourceDir, path), "utf8");
2208
- files.set(path, text.replaceAll("{{version}}", version2));
2484
+ files.set(
2485
+ path,
2486
+ text.replaceAll("{{version}}", version2).replaceAll("{{exportColumns}}", exportColumnsMarkdown())
2487
+ );
2209
2488
  }
2210
2489
  return files;
2211
2490
  }
@@ -2435,7 +2714,7 @@ function tilde(path, home) {
2435
2714
  }
2436
2715
  function printEntries(out, result, home) {
2437
2716
  for (const { dir, path, action } of result.entries) {
2438
- const line = `${SERVES[dir]}: ${ACTION_TEXT[action]} ${kleur6.gray(tilde(path, home))}`;
2717
+ const line = `${SERVES[dir]}: ${ACTION_TEXT[action]} ${kleur7.gray(tilde(path, home))}`;
2439
2718
  if (action.startsWith("skipped")) out.warn(line);
2440
2719
  else out.success(line);
2441
2720
  }
@@ -2455,17 +2734,17 @@ function registerSkillsCommands(program, deps2, version2) {
2455
2734
  const result = await installSkill(paths(), deps2.skillSource, version2, agents, { force: opts.force });
2456
2735
  if (out.json) return out.data({ version: version2, agents, ...result });
2457
2736
  out.info(
2458
- `${kleur6.bold("planttaxomatcher")} skill ${version2} \u2192 ${kleur6.gray(tilde(result.canonical, deps2.home))}`
2737
+ `${kleur7.bold("planttaxomatcher")} skill ${version2} \u2192 ${kleur7.gray(tilde(result.canonical, deps2.home))}`
2459
2738
  );
2460
2739
  printEntries(out, result, deps2.home);
2461
- for (const agent of agents) if (RELOAD_HINTS[agent]) out.info(kleur6.gray(RELOAD_HINTS[agent]));
2462
- out.info(kleur6.gray(`Try it: ask your agent to "match the plant names in my CSV with PlantTaxoMatcher".`));
2740
+ for (const agent of agents) if (RELOAD_HINTS[agent]) out.info(kleur7.gray(RELOAD_HINTS[agent]));
2741
+ out.info(kleur7.gray(`Try it: ask your agent to "match the plant names in my CSV with PlantTaxoMatcher".`));
2463
2742
  });
2464
2743
  skills.command("uninstall").description("Remove the links and copies this CLI installed, then its own copy of the skill").option("--force", "Also remove copies edited by hand", false).option("--json", "Print the result as JSON", false).action(async (opts) => {
2465
2744
  const out = createOutput(deps2.stdout, deps2.stderr, opts.json, deps2.now);
2466
2745
  const result = await uninstallSkill(paths(), { force: opts.force });
2467
2746
  if (out.json) return out.data(result);
2468
- if (result.entries.length === 0) out.info(kleur6.gray("No installed skill links found."));
2747
+ if (result.entries.length === 0) out.info(kleur7.gray("No installed skill links found."));
2469
2748
  printEntries(out, result, deps2.home);
2470
2749
  });
2471
2750
  skills.command("status").description("Show where the skill is installed and which copy each agent loads").option("--json", "Print the status as JSON", false).action(async (opts) => {
@@ -2474,28 +2753,28 @@ function registerSkillsCommands(program, deps2, version2) {
2474
2753
  if (out.json) return out.data({ cliVersion: version2, ...status });
2475
2754
  const { canonical } = status;
2476
2755
  out.info(
2477
- canonical.version ? `Skill ${canonical.version}${canonical.pristine ? "" : " (edited)"} at ${kleur6.gray(tilde(canonical.path, deps2.home))}` : kleur6.gray("Not installed. Run `planttaxomatcher skills install`.")
2756
+ canonical.version ? `Skill ${canonical.version}${canonical.pristine ? "" : " (edited)"} at ${kleur7.gray(tilde(canonical.path, deps2.home))}` : kleur7.gray("Not installed. Run `planttaxomatcher skills install`.")
2478
2757
  );
2479
2758
  for (const entry of status.entries) {
2480
2759
  if (entry.kind !== "missing")
2481
- out.info(` ${entry.kind.padEnd(8)} ${kleur6.gray(tilde(entry.path, deps2.home))}`);
2760
+ out.info(` ${entry.kind.padEnd(8)} ${kleur7.gray(tilde(entry.path, deps2.home))}`);
2482
2761
  }
2483
2762
  for (const { agent, installed, loads, state } of status.agents) {
2484
2763
  const entry = status.entries.find((e) => e.dir === loads);
2485
2764
  const where = entry ? `${state} in ${tilde(entry.path, deps2.home)}` : "no skill";
2486
- out.info(` ${AGENT_NAMES[agent].padEnd(12)} ${installed ? where : kleur6.gray(`not found (${where})`)}`);
2765
+ out.info(` ${AGENT_NAMES[agent].padEnd(12)} ${installed ? where : kleur7.gray(`not found (${where})`)}`);
2487
2766
  }
2488
2767
  });
2489
2768
  }
2490
2769
 
2491
2770
  // src/commands/submit.ts
2492
2771
  import { parse as parsePath } from "path";
2493
- import kleur8 from "kleur";
2772
+ import kleur9 from "kleur";
2494
2773
 
2495
2774
  // src/dry-run.ts
2496
2775
  import { promises as fs4, createReadStream } from "fs";
2497
2776
  import Papa from "papaparse";
2498
- import kleur7 from "kleur";
2777
+ import kleur8 from "kleur";
2499
2778
  async function readSample(file, sampleLimit) {
2500
2779
  const lower = file.toLowerCase();
2501
2780
  if (lower.endsWith(".json")) {
@@ -2600,23 +2879,23 @@ async function buildDryRunReport(file, opts) {
2600
2879
  }
2601
2880
  function printDryRunReport(report, log) {
2602
2881
  log("");
2603
- log(kleur7.bold("Dry-run preview"));
2882
+ log(kleur8.bold("Dry-run preview"));
2604
2883
  log(
2605
- ` Read ${kleur7.cyan(report.rowsRead)} rows \xB7 ${kleur7.yellow(report.nullNameCount)} with no usable name \xB7 ${kleur7.yellow(report.qualifierFlagCount)} with qualifier flags`
2884
+ ` Read ${kleur8.cyan(report.rowsRead)} rows \xB7 ${kleur8.yellow(report.nullNameCount)} with no usable name \xB7 ${kleur8.yellow(report.qualifierFlagCount)} with qualifier flags`
2606
2885
  );
2607
2886
  if (report.idTypeDetection) {
2608
2887
  const d = report.idTypeDetection;
2609
2888
  const pct = Math.round(d.dominantConfidence * 100);
2610
- log(` ID-column detection: ${kleur7.green(d.dominant ?? "\u2014")} (${pct}% of ${d.sampleSize} sampled)`);
2889
+ log(` ID-column detection: ${kleur8.green(d.dominant ?? "\u2014")} (${pct}% of ${d.sampleSize} sampled)`);
2611
2890
  if (d.minorityExamples.length > 0) {
2612
2891
  log(` Minority examples:`);
2613
2892
  for (const m of d.minorityExamples) {
2614
- log(` row ${m.rowIndex + 1}: ${kleur7.gray(m.value)} \u2192 ${kleur7.dim(m.type)}`);
2893
+ log(` row ${m.rowIndex + 1}: ${kleur8.gray(m.value)} \u2192 ${kleur8.dim(m.type)}`);
2615
2894
  }
2616
2895
  }
2617
2896
  }
2618
2897
  log("");
2619
- log(kleur7.bold("First rows after normalization:"));
2898
+ log(kleur8.bold("First rows after normalization:"));
2620
2899
  log(` ${"#".padStart(4)} ${"input".padEnd(36)} ${"normalized".padEnd(36)} flags`);
2621
2900
  for (const r of report.sampleRows) {
2622
2901
  const idx = String(r.rowIndex + 1).padStart(4);
@@ -2677,7 +2956,9 @@ function buildJobConfig(opts) {
2677
2956
  speciesLevelAcceptedOnly: opts.speciesLevel,
2678
2957
  acceptUnconfirmedAuthor: opts.ignoreAuthor,
2679
2958
  exportConfirmedOnly: false,
2680
- ...opts.referential ? { referentialVersion: opts.referential } : {}
2959
+ referential: parseBackbone(opts.backbone),
2960
+ ...opts.referential ? { referentialVersion: opts.referential } : {},
2961
+ ...opts.area?.trim() ? { area: opts.area.trim() } : {}
2681
2962
  };
2682
2963
  const checked = jobConfigSchema.safeParse(config);
2683
2964
  if (!checked.success) {
@@ -2700,13 +2981,20 @@ function summarize(submitted, finals, creds) {
2700
2981
  url: jobUrl(creds, job.id)
2701
2982
  }));
2702
2983
  }
2984
+ async function checkAgainstServer(deps2, creds, config) {
2985
+ const backbone = config.referential ?? "wcvp";
2986
+ if (config.referentialVersion) {
2987
+ requireVersion(await fetchVersions(deps2, creds, backbone), backbone, config.referentialVersion);
2988
+ }
2989
+ if (config.area) config.area = requireArea(await deps2.api.listAreas(creds), config.area).code;
2990
+ }
2703
2991
  function jobNameFor(file, fileCount, name) {
2704
2992
  return fileCount === 1 && name?.trim() ? name.trim() : parsePath(file).name;
2705
2993
  }
2706
2994
  async function confirmDryRun(deps2, out, files, opts) {
2707
2995
  const previewRows = Number(opts.dryRunRows);
2708
2996
  for (const file of files) {
2709
- if (files.length > 1) out.info(kleur8.bold(`
2997
+ if (files.length > 1) out.info(kleur9.bold(`
2710
2998
  ${file}`));
2711
2999
  const report = await buildDryRunReport(file, {
2712
3000
  nameColumn: opts.nameColumn,
@@ -2741,13 +3029,17 @@ function registerSubmitCommand(program, deps2) {
2741
3029
  "--ignore-author",
2742
3030
  "Accept a unique canonical name match whose author couldn't be confirmed instead of sending it to review (a CONFLICTING author still reviews)",
2743
3031
  false
2744
- ).option("--keep-infraspecific", "Deprecated \u2014 now the default; has no effect", false).option("--allow-llm", "Allow the Layer 7 LLM to break ties between competing matches", false).option("--llm-cap-cents <cents>", "Max LLM spend in cents", "500").option(
3032
+ ).option("--keep-infraspecific", "Deprecated \u2014 now the default; has no effect", false).option("--allow-llm", "Allow the Layer 7 LLM to break ties between competing matches", false).option("--llm-cap-cents <cents>", "Max LLM spend in cents", "500").option("--backbone <name>", "Backbone to match against: wcvp or wfo", "wcvp").option(
2745
3033
  "--referential <version>",
2746
- "WCVP snapshot version to match against (default: your team's default snapshot, else the newest import)"
3034
+ "Backbone version to match against (default: the one `planttaxomatcher versions` marks as default)"
3035
+ ).option(
3036
+ "--area <code>",
3037
+ "WGSRPD level-3 area (botanical country, e.g. FRA): an ambiguous match is narrowed to the taxa native there. See `planttaxomatcher areas`"
2747
3038
  ).option("--no-watch", "Return as soon as the jobs are created").option("--dry-run", "Preview the first rows locally, then confirm before uploading", false).option("--dry-run-rows <n>", "Rows to show in the dry-run preview", "10").option("-y, --yes", "Skip the --dry-run confirmation (needed without a terminal)", false).option("--json", "Print the created jobs as a JSON array (with --watch: once they end)", false).action(async (patterns, opts) => {
2748
3039
  const out = createOutput(deps2.stdout, deps2.stderr, opts.json, deps2.now);
2749
3040
  const config = buildJobConfig(opts);
2750
3041
  const creds = await requireCredentials(deps2.store, deps2.env);
3042
+ await checkAgainstServer(deps2, creds, config);
2751
3043
  const files = await expandInputs(patterns);
2752
3044
  if (opts.name && files.length > 1)
2753
3045
  out.warn("--name ignored for a multi-file submit; each job is named after its file");
@@ -2757,11 +3049,11 @@ function registerSubmitCommand(program, deps2) {
2757
3049
  );
2758
3050
  }
2759
3051
  if (files.length > 1) {
2760
- out.info(`${kleur8.cyan("\u2192")} ${files.length} files matched:`);
2761
- for (const file of files) out.info(kleur8.gray(` ${file}`));
3052
+ out.info(`${kleur9.cyan("\u2192")} ${files.length} files matched:`);
3053
+ for (const file of files) out.info(kleur9.gray(` ${file}`));
2762
3054
  }
2763
3055
  if (opts.dryRun && !await confirmDryRun(deps2, out, files, opts)) {
2764
- out.info(kleur8.gray("Aborted."));
3056
+ out.info(kleur9.gray("Aborted."));
2765
3057
  return;
2766
3058
  }
2767
3059
  const submitted = [];
@@ -2769,16 +3061,16 @@ function registerSubmitCommand(program, deps2) {
2769
3061
  try {
2770
3062
  for (const file of files) {
2771
3063
  const name = jobNameFor(file, files.length, opts.name);
2772
- out.info(`${kleur8.cyan("\u2192")} Uploading ${file} \u2026`);
3064
+ out.info(`${kleur9.cyan("\u2192")} Uploading ${file} \u2026`);
2773
3065
  const job = await deps2.api.submitJob(creds, file, config, name || null);
2774
- out.success(`Job created: ${job.id} ${kleur8.gray(name)}`);
3066
+ out.success(`Job created: ${job.id} ${kleur9.gray(name)}`);
2775
3067
  out.info(` rows=${job.totalRows} uniqueQueries=${job.uniqueQueries}`);
2776
3068
  submitted.push({ file, job });
2777
3069
  }
2778
3070
  if (opts.watch) {
2779
3071
  const watchOut = opts.json ? createOutput(deps2.stderr, deps2.stderr, false, deps2.now) : out;
2780
3072
  for (const { file, job } of submitted) {
2781
- if (submitted.length > 1) watchOut.info(kleur8.bold(`
3073
+ if (submitted.length > 1) watchOut.info(kleur9.bold(`
2782
3074
  [${file}] ${job.id}`));
2783
3075
  finals.push({ id: job.id, status: await watchJob(deps2.api, watchOut, creds, job.id) });
2784
3076
  }
@@ -2813,6 +3105,7 @@ function buildProgram(deps2, version2) {
2813
3105
  registerAuthCommands(program, deps2);
2814
3106
  registerJobCommands(program, deps2);
2815
3107
  registerSubmitCommand(program, deps2);
3108
+ registerCatalogCommands(program, deps2);
2816
3109
  registerDownloadCommand(program, deps2);
2817
3110
  registerSkillsCommands(program, deps2, version2);
2818
3111
  return program;
@@ -2823,9 +3116,9 @@ async function runCli(argv, deps2, version2) {
2823
3116
  return 0;
2824
3117
  } catch (err) {
2825
3118
  const { exitCode, message, hint } = describeError(err);
2826
- if (message) deps2.stderr.write(`${kleur9.red("error:")} ${message}
3119
+ if (message) deps2.stderr.write(`${kleur10.red("error:")} ${message}
2827
3120
  `);
2828
- if (hint) deps2.stderr.write(`${kleur9.gray(hint)}
3121
+ if (hint) deps2.stderr.write(`${kleur10.gray(hint)}
2829
3122
  `);
2830
3123
  return exitCode;
2831
3124
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@plantnet/planttaxomatcher",
3
- "version": "0.3.0",
3
+ "version": "0.4.0",
4
4
  "description": "PlantTaxoMatcher CLI — reconcile plant names against WCVP via the PlantTaxoMatcher API.",
5
5
  "type": "module",
6
6
  "engines": {
@@ -62,9 +62,21 @@ row of an XLSX sheet) and choose the columns:
62
62
  - `--family-column`, `--genus-column`, `--rank-column` when present: they help break ties.
63
63
  - `--id-column` when rows already carry an identifier, with `--id-type wcvp` or `--id-type gbif` if you know which.
64
64
 
65
+ Then the reference to match against:
66
+
67
+ - **Backbone**: WCVP by default. Pass `--backbone wfo` only when the user asks for World Flora Online.
68
+ - **Version**: leave `--referential` out to use the server's default.
69
+ `planttaxomatcher versions --json` lists the versions; the one with
70
+ `"isDefault": true` is used when none is named. Pass `--referential <version>`
71
+ only when the user names one, or needs a team-edited version.
72
+ - **Area**: when the user says where the plants grow (a country, a regional
73
+ flora), pass `--area <code>` so an ambiguous name is narrowed to the taxa
74
+ native there. Find the WGSRPD level-3 code with
75
+ `planttaxomatcher areas --search <country> --json`, e.g. `FRA` for France.
76
+
65
77
  Ask the user when the name column is not obvious. Keep the defaults for
66
78
  everything else unless the user asks; `references/submit-options.md` lists
67
- every option (author handling, species-level roll-up, snapshot, row filter, LLM).
79
+ every option (author handling, species-level roll-up, row filter, LLM).
68
80
 
69
81
  ## 3. Submit
70
82
 
@@ -93,11 +105,29 @@ planttaxomatcher download <jobId> --format csv --output results.csv --json
93
105
  ```
94
106
 
95
107
  The export keeps every column of the user's file and adds `planttaxomatcher_*`
96
- and `wcvp_*` columns; `references/results.md` explains them, the grades and the
97
- review states. Useful options: `--confirmed-only`, `--dedupe`,
98
- `--wcvp-extra ipni_id,powo_id`, `--format xlsx`, and `--bundle` for a ZIP with
99
- a NOTICE.md that cites the WCVP snapshot. `--list-columns --json` shows every
100
- available column.
108
+ and `wcvp_*` columns. The cleaned name to hand back is
109
+ **`wcvp_accepted_name_with_author`**: the accepted name with its author, ready
110
+ to cite (e.g. *Calicotome spinosa (L.) Link*), filled even when the input was a
111
+ synonym. `wcvp_accepted_name` is the same without the author, and
112
+ `wcvp_matched_name` is what the input matched before synonyms were resolved.
113
+
114
+ The filters of the web app's download dialog. Check what each keeps first
115
+ with `planttaxomatcher download <jobId> --counts --json`:
116
+
117
+ - `--confirmed-only`: only rows accepted automatically (grade A) or by a reviewer.
118
+ - `--resolved-only`: drop the rows with no accepted name (no match, error, rejected).
119
+ - `--dedupe`: one row per accepted taxon (synonyms and duplicates collapse).
120
+ - `--taxon-level species_below` (drop genus and family matches) or `--taxon-level species_only`.
121
+
122
+ To shape the file:
123
+
124
+ - `--columns` keeps only the listed columns. Start with the user's own name column (the one passed to `--name-column`), e.g. `--columns <name column>,wcvp_accepted_name_with_author,planttaxomatcher_grade,planttaxomatcher_review_status`.
125
+ - `--wcvp-extra ipni_id,powo_id` appends WCVP fields as `wcvp_<field>` columns.
126
+ - `--format xlsx`, or `--bundle` for a ZIP with a NOTICE.md that cites the backbone version.
127
+
128
+ `planttaxomatcher download <jobId> --list-columns --json` lists every column
129
+ with what it holds; `references/results.md` describes them all, with the grades
130
+ and review states.
101
131
 
102
132
  ## 6. Report back
103
133
 
@@ -105,7 +135,9 @@ Summarise from the status counts and the export:
105
135
 
106
136
  - rows matched and accepted automatically (grade A),
107
137
  - rows waiting for review (grade B or C, ambiguous),
108
- - rows with no match or an error, with a few examples.
138
+ - rows with no match or an error, with a few examples,
139
+
140
+ quoting names as input → `wcvp_accepted_name_with_author`.
109
141
 
110
142
  Give the job's `url`: reviewing happens in the web app, not the CLI. Call a
111
143
  name accepted only when `planttaxomatcher_review_status` is `not_required`,
@@ -114,4 +146,5 @@ name accepted only when `planttaxomatcher_review_status` is `not_required`,
114
146
  ## Other commands
115
147
 
116
148
  - `planttaxomatcher list --json`: recent jobs (`--status completed`, `--limit 50`).
149
+ - `planttaxomatcher versions --json` and `planttaxomatcher areas --json`: what `--referential` and `--area` accept.
117
150
  - `planttaxomatcher pause <jobId>`, `planttaxomatcher resume <jobId>`, `planttaxomatcher cancel <jobId>`: control a running job; each takes `--json`. Cancelling keeps the rows already matched. Ask before cancelling.
@@ -2,37 +2,20 @@
2
2
 
3
3
  Every row of the user's file comes back with its original columns, plus the
4
4
  columns below. `planttaxomatcher download <jobId> --list-columns --json` lists
5
- them all for a given job.
6
-
7
- ## Match columns
8
-
9
- | Column | Meaning |
10
- | ----------------------------------- | ---------------------------------------------------------------------------------------------------------------------------------------------------------------- |
11
- | `planttaxomatcher_match_status` | `matched`, `ambiguous` (several candidates), `no_match`, `error`, `skipped` (filtered out by `--filter-column`, or no usable name) |
12
- | `planttaxomatcher_grade` | `A`, `B` or `C` (see below); empty without a match |
13
- | `planttaxomatcher_review_status` | `not_required`, `pending`, `accepted`, `rejected`, `overridden` (see below) |
14
- | `planttaxomatcher_evidence_type` | How the match was found: `local_exact`, `local_canonical_unique`, `local_fuzzy`, `external_validated`, `external_fuzzy`, `team_history`, `cross_backbone`, `llm` |
15
- | `planttaxomatcher_confidence` | Score between 0 and 1 |
16
- | `planttaxomatcher_flags` | Why a match needs a look, e.g. `author-unconfirmed`, `author-mismatch` |
17
- | `planttaxomatcher_reason` | One-line explanation of the decision |
18
- | `planttaxomatcher_alternatives` | Other candidates, for ambiguous rows |
19
- | `planttaxomatcher_input_normalized` | The name as matched, after cleaning |
20
- | `planttaxomatcher_input_qualifier` | A qualifier found in the input (`cf`, `aff`, `sp`, `aggregate`…) |
21
-
22
- ## WCVP columns
23
-
24
- | Column | Meaning |
25
- | ------------------------------------------------------------------------------ | -------------------------------------------------- |
26
- | `wcvp_matched_name` | The WCVP name the input matched (may be a synonym) |
27
- | `wcvp_matched_taxonomic_status` | Its status in WCVP: `Accepted`, `Synonym`, … |
28
- | `wcvp_accepted_name`, `wcvp_accepted_author`, `wcvp_accepted_name_with_author` | The accepted name the match resolves to |
29
- | `wcvp_accepted_taxon_id` | WCVP id of the accepted taxon |
30
- | `wcvp_family`, `wcvp_rank` | Family and rank of the accepted taxon |
31
- | `ipni_lsid`, `powo_url` | Links to IPNI and Plants of the World Online |
32
-
33
- `--wcvp-extra` appends more WCVP fields as `wcvp_<key>`, for example
34
- `ipni_id`, `powo_id`, `geographic_area`, `lifeform_description`,
35
- `first_published`.
5
+ them for a given job (the Pl@ntNet columns only exist when the server has
6
+ Pl@ntNet enabled).
7
+
8
+ ## The columns that matter most
9
+
10
+ - `wcvp_accepted_name_with_author`: the cleaned name to hand back, the accepted
11
+ name with its author (e.g. *Calicotome spinosa (L.) Link*), even when the
12
+ input was a synonym.
13
+ - `planttaxomatcher_grade` and `planttaxomatcher_review_status`: how far to
14
+ trust it (below).
15
+ - `planttaxomatcher_flags` and `planttaxomatcher_reason`: why a row needs a look.
16
+ - `wcvp_accepted_taxon_id`, `ipni_lsid`, `powo_url`: identifiers and links for
17
+ the accepted taxon (WCVP jobs; a WFO job's id is in
18
+ `planttaxomatcher_accepted_identifier`).
36
19
 
37
20
  ## Grades
38
21
 
@@ -53,3 +36,5 @@ them all for a given job.
53
36
 
54
37
  Only `not_required`, `accepted` and `overridden` rows carry a name the user
55
38
  can rely on. `--confirmed-only` exports just those.
39
+
40
+ {{exportColumns}}
@@ -16,6 +16,16 @@ something else; they are what the web app uses.
16
16
  | `--id-type <type>` | What `--id-column` holds: `auto` (default), `wcvp`, `gbif` |
17
17
  | `--filter-column <name>` and `--filter-value <value>` | Only match rows whose column equals the value (trimmed, case-insensitive), e.g. `--filter-column kingdom --filter-value Plantae` |
18
18
 
19
+ ## Backbone, version and area
20
+
21
+ | Option | Default | Effect |
22
+ | --- | --- | --- |
23
+ | `--backbone <name>` | `wcvp` | `wcvp` (World Checklist of Vascular Plants) or `wfo` (World Flora Online) |
24
+ | `--referential <version>` | the server's default | Backbone version, e.g. `wcvp-v14`. `planttaxomatcher versions --json` lists them and marks the default (`isDefault`); a team-edited version is listed with `kind: derived` |
25
+ | `--area <code>` | none | WGSRPD level-3 area (botanical country), e.g. `FRA`: an ambiguous name is narrowed to the taxa native there. `planttaxomatcher areas --search <text> --json` finds the code |
26
+
27
+ The CLI checks `--referential` and `--area` against the server before uploading, and names the valid values on a typo.
28
+
19
29
  ## Matching
20
30
 
21
31
  | Option | Default | Effect |
@@ -23,7 +33,6 @@ something else; they are what the web app uses.
23
33
  | `--author-mode <mode>` | `prefer` | `ignore`, `prefer` or `strict` handling of authorship |
24
34
  | `--ignore-author` | off | Accept a unique name match whose author could not be confirmed, instead of sending it to review. A conflicting author still goes to review |
25
35
  | `--species-level` | off | Roll varieties, subspecies and forms up to their accepted species |
26
- | `--referential <version>` | team default | WCVP snapshot to match against, e.g. `wcvp-v14` |
27
36
  | `--allow-llm` | off | Let a language model break ties between competing matches |
28
37
  | `--llm-cap-cents <cents>` | `500` | Spending cap for `--allow-llm` |
29
38
  | `--review-mode <mode>` | `recommended` | `off`, `recommended` or `strict` |