@plantnet/planttaxomatcher 0.3.0 → 0.4.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.js
CHANGED
|
@@ -251,6 +251,8 @@ var wcvpSnapshotSummarySchema = z4.object({
|
|
|
251
251
|
importedAt: z4.string().datetime(),
|
|
252
252
|
/** Newest OFFICIAL snapshot (derived ones never count as latest). */
|
|
253
253
|
isLatest: z4.boolean(),
|
|
254
|
+
/** The version a job uses when none is named: the team's default if still visible, else the latest. */
|
|
255
|
+
isDefault: z4.boolean(),
|
|
254
256
|
kind: z4.enum(["official", "derived"]),
|
|
255
257
|
baseVersion: z4.string().nullable(),
|
|
256
258
|
label: z4.string().nullable(),
|
|
@@ -1046,6 +1048,8 @@ var wfoSnapshotSummarySchema = z8.object({
|
|
|
1046
1048
|
recordCount: z8.number().int().nonnegative(),
|
|
1047
1049
|
importedAt: z8.string().datetime(),
|
|
1048
1050
|
isLatest: z8.boolean(),
|
|
1051
|
+
/** The version a job uses when none is named: the team's default if still visible, else the latest. */
|
|
1052
|
+
isDefault: z8.boolean(),
|
|
1049
1053
|
kind: snapshotKindSchema,
|
|
1050
1054
|
baseVersion: z8.string().nullable(),
|
|
1051
1055
|
label: z8.string().nullable(),
|
|
@@ -1358,6 +1362,87 @@ function parseEnvFlag(raw, fallback) {
|
|
|
1358
1362
|
// ../shared/src/public-server.ts
|
|
1359
1363
|
var DEFAULT_SERVER_URL = "https://planttaxomatcher.plantnet.org";
|
|
1360
1364
|
|
|
1365
|
+
// ../shared/src/export-columns.ts
|
|
1366
|
+
import { z as z11 } from "zod";
|
|
1367
|
+
var RESULT_COLUMN_DESCRIPTIONS = {
|
|
1368
|
+
planttaxomatcher_job_id: "Id of the job this row belongs to.",
|
|
1369
|
+
planttaxomatcher_input_normalized: "The input name as it was matched, after cleaning (spacing, case, qualifiers removed).",
|
|
1370
|
+
planttaxomatcher_input_qualifier: "Identification qualifier found in the input: cf, aff, sensu_lato, sensu_stricto, aggregate, complex, sp, spp, indet, cultivar, hybrid_formula, informal. Empty when none.",
|
|
1371
|
+
planttaxomatcher_parse_quality: "How cleanly the name parsed (gnparser): 1 clean, 2 minor issues, 3 significant issues, 4 badly formed; empty when unparsed.",
|
|
1372
|
+
planttaxomatcher_match_status: "matched, ambiguous (several candidates, none chosen), no_match, error, or skipped (filtered out, or no usable name).",
|
|
1373
|
+
planttaxomatcher_evidence_type: "How the match was found: local_exact, local_canonical_unique, local_fuzzy, external_validated, external_fuzzy, team_history, cross_backbone, llm.",
|
|
1374
|
+
planttaxomatcher_grade: "A (confident, accepted automatically), B (plausible, needs review) or C (weakest: a language-model suggestion or genus only).",
|
|
1375
|
+
planttaxomatcher_review_status: "not_required (grade A), pending (waiting for review), accepted, rejected, or overridden (a reviewer picked another taxon).",
|
|
1376
|
+
planttaxomatcher_confidence: "Match score between 0 and 1.",
|
|
1377
|
+
planttaxomatcher_score_breakdown: "JSON of the score components: canonical, author, rank and family similarity, provider agreement.",
|
|
1378
|
+
planttaxomatcher_layer: "The cascade layer family that settled the row (same values as the evidence type).",
|
|
1379
|
+
planttaxomatcher_reason: "One-line explanation of the decision.",
|
|
1380
|
+
planttaxomatcher_flags: "Comma-separated reasons a match needs a look, e.g. author-unconfirmed, author-mismatch, genus-fallback.",
|
|
1381
|
+
planttaxomatcher_matched_identifier_namespace: "Namespace of the matched name\u2019s identifier, e.g. wcvp:taxonID or wfo:taxonID.",
|
|
1382
|
+
planttaxomatcher_matched_identifier: "Identifier of the matched name (possibly a synonym).",
|
|
1383
|
+
planttaxomatcher_accepted_identifier_namespace: "Namespace of the accepted taxon\u2019s identifier: wcvp:acceptedNameUsageID or wfo:acceptedNameUsageID.",
|
|
1384
|
+
planttaxomatcher_accepted_identifier: "Identifier of the accepted taxon, in the job\u2019s backbone (WCVP or WFO).",
|
|
1385
|
+
plantnet_identifiable: "true when Pl@ntNet can identify the accepted taxon from photos (when Pl@ntNet is enabled).",
|
|
1386
|
+
plantnet_id: "Pl@ntNet species id of the accepted taxon (when Pl@ntNet is enabled).",
|
|
1387
|
+
wcvp_matched_name: "The backbone name the input matched, which may be a synonym.",
|
|
1388
|
+
wcvp_matched_taxonomic_status: "Status of the matched name in the backbone, e.g. Accepted, Synonym or Illegitimate.",
|
|
1389
|
+
wcvp_accepted_name: "The accepted name the match resolves to, without author.",
|
|
1390
|
+
wcvp_accepted_author: "Author of the accepted name.",
|
|
1391
|
+
wcvp_accepted_name_with_author: "The accepted name and its author in one cell, ready to cite, e.g. Calicotome spinosa (L.) Link. The column to hand back as the cleaned name.",
|
|
1392
|
+
wcvp_accepted_taxon_id: "WCVP id (plant_name_id) of the accepted taxon. Empty for WFO jobs, whose id is in planttaxomatcher_accepted_identifier.",
|
|
1393
|
+
wcvp_accepted_usage_id: "Same id as wcvp_accepted_taxon_id, under its Darwin Core name (acceptedNameUsageID).",
|
|
1394
|
+
ipni_lsid: "IPNI LSID of the accepted name: urn:lsid:ipni.org:names: followed by its IPNI id.",
|
|
1395
|
+
powo_url: "Plants of the World Online page of the accepted taxon.",
|
|
1396
|
+
wcvp_family: "Family of the accepted taxon.",
|
|
1397
|
+
wcvp_rank: "Rank of the accepted taxon, e.g. Species, Subspecies, Variety or Genus.",
|
|
1398
|
+
planttaxomatcher_alternatives: "JSON list of the other candidates, mostly for ambiguous rows.",
|
|
1399
|
+
planttaxomatcher_reviewed: "true when a person reviewed the row.",
|
|
1400
|
+
planttaxomatcher_referential_version: "Backbone version the job matched against, e.g. wcvp-v14.",
|
|
1401
|
+
planttaxomatcher_normalizer_version: "Version of the name normaliser, for reproducibility.",
|
|
1402
|
+
planttaxomatcher_parser_version: "Version of the name parser, for reproducibility.",
|
|
1403
|
+
planttaxomatcher_matcher_version: "Version of the matching cascade, for reproducibility.",
|
|
1404
|
+
planttaxomatcher_scoring_version: "Version of the grading rules, for reproducibility."
|
|
1405
|
+
};
|
|
1406
|
+
var taxonLevelSchema = z11.enum(["all", "species_below", "species_only"]);
|
|
1407
|
+
var WCVP_EXTRA_DESCRIPTIONS = {
|
|
1408
|
+
ipni_id: "IPNI id of the name (the part after urn:lsid:ipni.org:names:).",
|
|
1409
|
+
powo_id: "Plants of the World Online id of the taxon.",
|
|
1410
|
+
plant_name_id: "WCVP id of the record.",
|
|
1411
|
+
accepted_plant_name_id: "WCVP id of the accepted name the record points to.",
|
|
1412
|
+
basionym_plant_name_id: "WCVP id of the basionym (the original name this one is based on).",
|
|
1413
|
+
parent_plant_name_id: "WCVP id of the parent taxon, e.g. the species of a variety.",
|
|
1414
|
+
taxon_authors: "Full author string of the name.",
|
|
1415
|
+
primary_author: "Author or authors who published the name.",
|
|
1416
|
+
parenthetical_author: "Author of the basionym, shown in parentheses in the author string.",
|
|
1417
|
+
publication_author: "Author of the book or article where the name was published, when different.",
|
|
1418
|
+
first_published: "Year of first publication.",
|
|
1419
|
+
place_of_publication: "Journal or book where the name was published.",
|
|
1420
|
+
volume_and_page: "Volume and page of the publication.",
|
|
1421
|
+
geographic_area: "Native distribution, as a short narrative statement.",
|
|
1422
|
+
climate_description: "Habitat or climate type, from published habitat information.",
|
|
1423
|
+
lifeform_description: "Life form, in a modified Raunki\xE6r system (e.g. phanerophyte).",
|
|
1424
|
+
homotypic_synonym: "Whether the name is a homotypic synonym (same type as the accepted name).",
|
|
1425
|
+
replaced_synonym_author: "Author of the replaced synonym, for a replacement name.",
|
|
1426
|
+
nomenclatural_remarks: "Nomenclatural notes, e.g. nom. illeg.",
|
|
1427
|
+
genus_hybrid: "Hybrid marker (\xD7) of the genus, when it is a hybrid.",
|
|
1428
|
+
species_hybrid: "Hybrid marker (\xD7) of the species, when it is a hybrid.",
|
|
1429
|
+
hybrid_formula: "Hybrid formula, e.g. Mentha aquatica \xD7 M. spicata.",
|
|
1430
|
+
infraspecies: "Infraspecific epithet.",
|
|
1431
|
+
infraspecific_rank: "Infraspecific rank, e.g. subsp., var. or f.",
|
|
1432
|
+
species: "Species epithet.",
|
|
1433
|
+
taxon_name: "Full name without author.",
|
|
1434
|
+
reviewed: "Whether Kew has peer-reviewed the family of this taxon.",
|
|
1435
|
+
pn_change: "Team-edited versions only: what the team changed on this record (codes such as re_accept;author).",
|
|
1436
|
+
pn_source: "Team-edited versions only: source of the change.",
|
|
1437
|
+
pn_taxon_id: "Team-edited versions only: the team\u2019s own taxon id.",
|
|
1438
|
+
wcvp13_taxon_status: "Team-edited versions only: taxonomic status before the edit (WCVP v13).",
|
|
1439
|
+
wcvp13_accepted_plant_name_id: "Team-edited versions only: accepted name id before the edit (WCVP v13).",
|
|
1440
|
+
wcvp13_taxon_name: "Team-edited versions only: name before the edit (WCVP v13).",
|
|
1441
|
+
wcvp13_taxon_authors: "Team-edited versions only: authors before the edit (WCVP v13).",
|
|
1442
|
+
wcvp13_family: "Team-edited versions only: family before the edit (WCVP v13).",
|
|
1443
|
+
wcvp13_genus: "Team-edited versions only: genus before the edit (WCVP v13)."
|
|
1444
|
+
};
|
|
1445
|
+
|
|
1361
1446
|
// src/errors.ts
|
|
1362
1447
|
import { CommanderError } from "commander";
|
|
1363
1448
|
var EXIT = {
|
|
@@ -1463,10 +1548,15 @@ function downloadFilename(contentDisposition, jobId, opts) {
|
|
|
1463
1548
|
if (suggested && suggested !== "." && suggested !== "..") return suggested;
|
|
1464
1549
|
return `planttaxomatcher_${jobId}.${opts.bundle ? "zip" : opts.format}`;
|
|
1465
1550
|
}
|
|
1551
|
+
function filterParams(filters, params = new URLSearchParams()) {
|
|
1552
|
+
if (filters.confirmedOnly) params.set("confirmedOnly", "true");
|
|
1553
|
+
if (filters.resolvedOnly) params.set("notEmptyOnly", "true");
|
|
1554
|
+
if (filters.dedupe) params.set("dedupe", "true");
|
|
1555
|
+
if (filters.taxonLevel && filters.taxonLevel !== "all") params.set("taxonLevel", filters.taxonLevel);
|
|
1556
|
+
return params;
|
|
1557
|
+
}
|
|
1466
1558
|
function downloadQuery(opts) {
|
|
1467
|
-
const params = new URLSearchParams({ format: opts.format });
|
|
1468
|
-
if (opts.confirmedOnly) params.set("confirmedOnly", "true");
|
|
1469
|
-
if (opts.dedupe) params.set("dedupe", "true");
|
|
1559
|
+
const params = filterParams(opts, new URLSearchParams({ format: opts.format }));
|
|
1470
1560
|
if (opts.bundle) params.set("bundle", "true");
|
|
1471
1561
|
if (opts.delimiter) params.set("delimiter", opts.delimiter);
|
|
1472
1562
|
if (opts.columns) params.set("columns", opts.columns);
|
|
@@ -1567,6 +1657,15 @@ function createApiClient(options = {}) {
|
|
|
1567
1657
|
},
|
|
1568
1658
|
getJob: (c, id) => call(c, `/v1/jobs/${encodeURIComponent(id)}`),
|
|
1569
1659
|
downloadColumns: (c, id) => call(c, `/v1/jobs/${encodeURIComponent(id)}/download/columns`),
|
|
1660
|
+
downloadCounts(c, id, filters) {
|
|
1661
|
+
const query = filterParams(filters);
|
|
1662
|
+
const suffix = query.size > 0 ? `?${query}` : "";
|
|
1663
|
+
return call(c, `/v1/jobs/${encodeURIComponent(id)}/download/count${suffix}`);
|
|
1664
|
+
},
|
|
1665
|
+
listVersions: (c, backbone) => call(c, `/v1/${backbone}/snapshots`),
|
|
1666
|
+
async listAreas(c) {
|
|
1667
|
+
return (await call(c, "/v1/wgsrpd/areas")).areas;
|
|
1668
|
+
},
|
|
1570
1669
|
pauseJob: (c, id) => call(c, `/v1/jobs/${encodeURIComponent(id)}/pause`, { method: "POST" }),
|
|
1571
1670
|
resumeJob: (c, id) => call(c, `/v1/jobs/${encodeURIComponent(id)}/resume`, { method: "POST" }),
|
|
1572
1671
|
cancelJob: (c, id) => call(c, `/v1/jobs/${encodeURIComponent(id)}/cancel`, { method: "POST" }),
|
|
@@ -1805,7 +1904,7 @@ function defaultDeps() {
|
|
|
1805
1904
|
|
|
1806
1905
|
// src/program.ts
|
|
1807
1906
|
import { Command } from "commander";
|
|
1808
|
-
import
|
|
1907
|
+
import kleur10 from "kleur";
|
|
1809
1908
|
|
|
1810
1909
|
// src/output.ts
|
|
1811
1910
|
import kleur from "kleur";
|
|
@@ -1980,11 +2079,124 @@ function registerAuthCommands(program, deps2) {
|
|
|
1980
2079
|
});
|
|
1981
2080
|
}
|
|
1982
2081
|
|
|
2082
|
+
// src/commands/catalog.ts
|
|
2083
|
+
import kleur3 from "kleur";
|
|
2084
|
+
|
|
2085
|
+
// src/catalog.ts
|
|
2086
|
+
function parseBackbone(value) {
|
|
2087
|
+
const parsed = backboneSchema.safeParse(value.trim().toLowerCase());
|
|
2088
|
+
if (!parsed.success) throw new CliError(`--backbone must be wcvp or wfo (got "${value}")`, EXIT.usage);
|
|
2089
|
+
return parsed.data;
|
|
2090
|
+
}
|
|
2091
|
+
function toVersionEntries(backbone, summaries) {
|
|
2092
|
+
return summaries.map((s) => ({
|
|
2093
|
+
backbone,
|
|
2094
|
+
version: s.version,
|
|
2095
|
+
kind: s.kind,
|
|
2096
|
+
label: s.label,
|
|
2097
|
+
baseVersion: s.baseVersion,
|
|
2098
|
+
isLatest: s.isLatest,
|
|
2099
|
+
// Servers before isDefault fall back to "latest", which is what they used.
|
|
2100
|
+
isDefault: s.isDefault ?? s.isLatest,
|
|
2101
|
+
recordCount: s.recordCount,
|
|
2102
|
+
importedAt: s.importedAt
|
|
2103
|
+
}));
|
|
2104
|
+
}
|
|
2105
|
+
var MAX_LISTED = 12;
|
|
2106
|
+
function requireVersion(entries, backbone, requested) {
|
|
2107
|
+
const found = entries.find((e) => e.version === requested);
|
|
2108
|
+
if (found) return found;
|
|
2109
|
+
const available = entries.map((e) => e.version);
|
|
2110
|
+
const listed = available.slice(0, MAX_LISTED).join(", ") + (available.length > MAX_LISTED ? ", \u2026" : "");
|
|
2111
|
+
throw new CliError(
|
|
2112
|
+
`no ${backbone.toUpperCase()} version "${requested}"${available.length ? ` (available: ${listed})` : ""}`,
|
|
2113
|
+
EXIT.usage,
|
|
2114
|
+
`Run \`planttaxomatcher versions --backbone ${backbone}\` to see them all.`
|
|
2115
|
+
);
|
|
2116
|
+
}
|
|
2117
|
+
function matches(area, needle) {
|
|
2118
|
+
return [area.code, area.name, area.region, area.continent].some((field) => field?.toLowerCase().includes(needle));
|
|
2119
|
+
}
|
|
2120
|
+
function searchAreas(areas, search) {
|
|
2121
|
+
const needle = search?.trim().toLowerCase();
|
|
2122
|
+
return needle ? areas.filter((area) => matches(area, needle)) : areas;
|
|
2123
|
+
}
|
|
2124
|
+
function requireArea(areas, requested) {
|
|
2125
|
+
const code = requested.trim().toUpperCase();
|
|
2126
|
+
const found = areas.find((a) => a.code.toUpperCase() === code);
|
|
2127
|
+
if (found) return found;
|
|
2128
|
+
const suggestions = searchAreas(areas, requested).slice(0, 5).map((a) => `${a.code} (${a.name ?? "?"})`);
|
|
2129
|
+
throw new CliError(
|
|
2130
|
+
`unknown area "${requested}"${suggestions.length ? `; did you mean ${suggestions.join(", ")}?` : ""}`,
|
|
2131
|
+
EXIT.usage,
|
|
2132
|
+
`--area takes a WGSRPD level-3 code: \`planttaxomatcher areas --search ${requested.trim()}\`.`
|
|
2133
|
+
);
|
|
2134
|
+
}
|
|
2135
|
+
function describeVersion(e) {
|
|
2136
|
+
const tags = [e.isDefault ? "default" : null, e.isLatest ? "latest" : null].filter(Boolean).join(", ");
|
|
2137
|
+
const origin = e.kind === "derived" ? `team-edited from ${e.baseVersion ?? "?"}` : "official";
|
|
2138
|
+
const label = e.label ? ` \u201C${e.label}\u201D` : "";
|
|
2139
|
+
return ` ${e.version.padEnd(28)} ${origin}${label}${tags ? ` [${tags}]` : ""} ${e.recordCount.toLocaleString("en")} names, imported ${e.importedAt.slice(0, 10)}`;
|
|
2140
|
+
}
|
|
2141
|
+
function versionLines(entries) {
|
|
2142
|
+
const lines = [];
|
|
2143
|
+
for (const backbone of ["wcvp", "wfo"]) {
|
|
2144
|
+
const ofBackbone = entries.filter((e) => e.backbone === backbone);
|
|
2145
|
+
if (ofBackbone.length === 0) continue;
|
|
2146
|
+
lines.push(backbone.toUpperCase(), ...ofBackbone.map(describeVersion));
|
|
2147
|
+
}
|
|
2148
|
+
return lines;
|
|
2149
|
+
}
|
|
2150
|
+
function areaLine(area) {
|
|
2151
|
+
const place2 = [area.region, area.continent].filter(Boolean).join(", ");
|
|
2152
|
+
return `${area.code.padEnd(4)} ${area.name ?? ""}${place2 ? ` (${place2})` : ""}`;
|
|
2153
|
+
}
|
|
2154
|
+
|
|
2155
|
+
// src/commands/catalog.ts
|
|
2156
|
+
async function fetchVersions(deps2, creds, backbone) {
|
|
2157
|
+
return toVersionEntries(backbone, await deps2.api.listVersions(creds, backbone));
|
|
2158
|
+
}
|
|
2159
|
+
function registerCatalogCommands(program, deps2) {
|
|
2160
|
+
program.command("versions").description(
|
|
2161
|
+
"List the backbone versions a job can match against (--referential), marking the one used by default"
|
|
2162
|
+
).option("--backbone <name>", "Only this backbone: wcvp or wfo (default: both)").option("--json", "Print the versions as a JSON array", false).action(async (opts) => {
|
|
2163
|
+
const out = createOutput(deps2.stdout, deps2.stderr, opts.json, deps2.now);
|
|
2164
|
+
const backbones = opts.backbone ? [parseBackbone(opts.backbone)] : backboneSchema.options;
|
|
2165
|
+
const creds = await requireCredentials(deps2.store, deps2.env);
|
|
2166
|
+
const entries = (await Promise.all(backbones.map((b) => fetchVersions(deps2, creds, b)))).flat();
|
|
2167
|
+
if (out.json) return out.data(entries);
|
|
2168
|
+
if (entries.length === 0) return out.info(kleur3.gray("No backbone version is installed on this server."));
|
|
2169
|
+
for (const line of versionLines(entries)) out.info(line);
|
|
2170
|
+
});
|
|
2171
|
+
program.command("areas").description("List the WGSRPD level-3 areas (botanical countries) that --area accepts").option("--search <text>", "Only areas whose code, name, region or continent contains this text").option("--json", "Print the areas as a JSON array", false).action(async (opts) => {
|
|
2172
|
+
const out = createOutput(deps2.stdout, deps2.stderr, opts.json, deps2.now);
|
|
2173
|
+
const creds = await requireCredentials(deps2.store, deps2.env);
|
|
2174
|
+
const areas = searchAreas(await deps2.api.listAreas(creds), opts.search);
|
|
2175
|
+
if (out.json) return out.data(areas);
|
|
2176
|
+
if (areas.length === 0) return out.info(kleur3.gray("No matching area."));
|
|
2177
|
+
for (const area of areas) out.info(areaLine(area));
|
|
2178
|
+
});
|
|
2179
|
+
}
|
|
2180
|
+
|
|
1983
2181
|
// src/commands/download.ts
|
|
1984
2182
|
import { writeFile } from "fs/promises";
|
|
1985
|
-
import
|
|
2183
|
+
import kleur4 from "kleur";
|
|
1986
2184
|
var FORMATS = ["csv", "xlsx", "json", "ndjson"];
|
|
1987
2185
|
var DELIMITERS = ["comma", "semicolon", "tab", "pipe"];
|
|
2186
|
+
var TAXON_LEVELS = taxonLevelSchema.options;
|
|
2187
|
+
function toDownloadFilters(opts) {
|
|
2188
|
+
const parsed = taxonLevelSchema.safeParse(opts.taxonLevel);
|
|
2189
|
+
if (!parsed.success) {
|
|
2190
|
+
throw new CliError(`--taxon-level must be ${TAXON_LEVELS.join(", ")} (got ${opts.taxonLevel})`, EXIT.usage);
|
|
2191
|
+
}
|
|
2192
|
+
const taxonLevel = parsed.data;
|
|
2193
|
+
return {
|
|
2194
|
+
confirmedOnly: opts.confirmedOnly,
|
|
2195
|
+
resolvedOnly: opts.resolvedOnly,
|
|
2196
|
+
dedupe: opts.dedupe,
|
|
2197
|
+
...taxonLevel !== "all" ? { taxonLevel } : {}
|
|
2198
|
+
};
|
|
2199
|
+
}
|
|
1988
2200
|
function toDownloadOptions(opts) {
|
|
1989
2201
|
const format = opts.format;
|
|
1990
2202
|
if (!FORMATS.includes(format)) {
|
|
@@ -1996,43 +2208,90 @@ function toDownloadOptions(opts) {
|
|
|
1996
2208
|
}
|
|
1997
2209
|
return {
|
|
1998
2210
|
format,
|
|
1999
|
-
|
|
2000
|
-
dedupe: opts.dedupe,
|
|
2211
|
+
...toDownloadFilters(opts),
|
|
2001
2212
|
bundle: opts.bundle,
|
|
2002
2213
|
...format === "csv" && delimiter !== "comma" ? { delimiter } : {},
|
|
2003
2214
|
...opts.columns ? { columns: opts.columns } : {},
|
|
2004
2215
|
...opts.wcvpExtra ? { wcvpExtra: opts.wcvpExtra } : {}
|
|
2005
2216
|
};
|
|
2006
2217
|
}
|
|
2218
|
+
function describeColumns(columns) {
|
|
2219
|
+
return {
|
|
2220
|
+
result: columns.result.map((c) => ({
|
|
2221
|
+
...c,
|
|
2222
|
+
description: c.description ?? RESULT_COLUMN_DESCRIPTIONS[c.key] ?? null
|
|
2223
|
+
})),
|
|
2224
|
+
original: columns.original,
|
|
2225
|
+
wcvpExtra: columns.wcvpExtra.map((f) => ({
|
|
2226
|
+
...f,
|
|
2227
|
+
description: f.description ?? WCVP_EXTRA_DESCRIPTIONS[f.key] ?? null
|
|
2228
|
+
}))
|
|
2229
|
+
};
|
|
2230
|
+
}
|
|
2007
2231
|
function printColumns(out, columns) {
|
|
2008
|
-
out.info(
|
|
2009
|
-
|
|
2232
|
+
const line = (key, description) => out.info(` ${key}${description ? `
|
|
2233
|
+
${kleur4.gray(description)}` : ""}`);
|
|
2234
|
+
out.info(kleur4.bold("Result columns (--columns):"));
|
|
2235
|
+
for (const column of columns.result) line(column.key, column.description);
|
|
2010
2236
|
if (columns.original.length > 0) {
|
|
2011
|
-
out.info(
|
|
2012
|
-
for (const key of columns.original)
|
|
2237
|
+
out.info(kleur4.bold("\nYour upload columns (--columns):"));
|
|
2238
|
+
for (const key of columns.original) line(key);
|
|
2239
|
+
}
|
|
2240
|
+
out.info(kleur4.bold("\nWCVP fields to append (--wcvp-extra), exported as wcvp_<field>:"));
|
|
2241
|
+
for (const field of columns.wcvpExtra) line(field.key, field.description);
|
|
2242
|
+
}
|
|
2243
|
+
function printCounts(out, counts) {
|
|
2244
|
+
const rows = [
|
|
2245
|
+
["all rows", counts.total, ""],
|
|
2246
|
+
["confirmed only", counts.confirmed, "--confirmed-only"],
|
|
2247
|
+
["resolved only", counts.resolved, "--resolved-only"],
|
|
2248
|
+
["one row per accepted taxon", counts.deduped, "--dedupe"],
|
|
2249
|
+
["with the filters given", counts.selected, ""]
|
|
2250
|
+
];
|
|
2251
|
+
for (const [label, count, flag] of rows) {
|
|
2252
|
+
out.info(` ${label.padEnd(28)} ${String(count).padStart(8)} ${kleur4.gray(flag)}`);
|
|
2013
2253
|
}
|
|
2014
|
-
out.info(kleur3.bold("\nWCVP extra fields (--wcvp-extra):"));
|
|
2015
|
-
for (const field of columns.wcvpExtra) out.info(` ${field.key} ${kleur3.gray(`(${field.group})`)}`);
|
|
2016
2254
|
}
|
|
2017
2255
|
function registerDownloadCommand(program, deps2) {
|
|
2018
|
-
program.command("download <jobId>").description("Download a job export (CSV, XLSX, JSON or NDJSON; optionally zipped with NOTICE.md)").option("--format <fmt>", FORMATS.join(" | "), "csv").option(
|
|
2256
|
+
program.command("download <jobId>").description("Download a job export (CSV, XLSX, JSON or NDJSON; optionally zipped with NOTICE.md)").option("--format <fmt>", FORMATS.join(" | "), "csv").option(
|
|
2257
|
+
"--confirmed-only",
|
|
2258
|
+
"Confirmed rows only: accepted automatically (grade A) or by a reviewer, or overridden",
|
|
2259
|
+
false
|
|
2260
|
+
).option(
|
|
2261
|
+
"--resolved-only",
|
|
2262
|
+
"Resolved rows only: drop rows with no accepted name (no match, error, rejected)",
|
|
2263
|
+
false
|
|
2264
|
+
).option(
|
|
2265
|
+
"--dedupe",
|
|
2266
|
+
"Remove duplicate accepted taxa: rows resolving to the same accepted taxon (synonyms, subspecies\u2026) become one line; unmatched rows are kept",
|
|
2267
|
+
false
|
|
2268
|
+
).option(
|
|
2269
|
+
"--taxon-level <level>",
|
|
2270
|
+
"Keep rows whose accepted taxon is at this level: all, species_below (species and infraspecific, drops genus/family), species_only. Unmatched rows are kept; add --resolved-only to drop them",
|
|
2271
|
+
"all"
|
|
2272
|
+
).option("--delimiter <sep>", `CSV separator: ${DELIMITERS.join(" | ")} (csv only)`, "comma").option("--bundle", "Wrap in a ZIP with NOTICE.md citing the WCVP snapshot and providers", false).option(
|
|
2019
2273
|
"--columns <list>",
|
|
2020
2274
|
"Comma-separated result/upload column keys to keep (default: all). See --list-columns"
|
|
2021
2275
|
).option(
|
|
2022
2276
|
"--wcvp-extra <list>",
|
|
2023
2277
|
"Comma-separated extra WCVP fields appended as wcvp_<key> columns (e.g. ipni_id,powo_id). See --list-columns"
|
|
2024
|
-
).option("--list-columns", "Print the columns available for this job and exit", false).option(
|
|
2278
|
+
).option("--list-columns", "Print the columns available for this job, with what each holds, and exit", false).option("--counts", "Print how many rows each filter keeps (as the web download dialog shows) and exit", false).option(
|
|
2025
2279
|
"--output <path>",
|
|
2026
2280
|
"Write to this path (default: the server-provided file name in the current directory)"
|
|
2027
|
-
).option("--json", "Print the result (or --list-columns) as JSON", false).action(async (jobId, opts) => {
|
|
2281
|
+
).option("--json", "Print the result (or --list-columns / --counts) as JSON", false).action(async (jobId, opts) => {
|
|
2028
2282
|
const out = createOutput(deps2.stdout, deps2.stderr, opts.json, deps2.now);
|
|
2029
2283
|
const download = toDownloadOptions(opts);
|
|
2030
2284
|
const creds = await requireCredentials(deps2.store, deps2.env);
|
|
2031
2285
|
if (opts.listColumns) {
|
|
2032
|
-
const columns = await deps2.api.downloadColumns(creds, jobId);
|
|
2286
|
+
const columns = describeColumns(await deps2.api.downloadColumns(creds, jobId));
|
|
2033
2287
|
if (out.json) return out.data(columns);
|
|
2034
2288
|
return printColumns(out, columns);
|
|
2035
2289
|
}
|
|
2290
|
+
if (opts.counts) {
|
|
2291
|
+
const counts = await deps2.api.downloadCounts(creds, jobId, toDownloadFilters(opts));
|
|
2292
|
+
if (out.json) return out.data(counts);
|
|
2293
|
+
return printCounts(out, counts);
|
|
2294
|
+
}
|
|
2036
2295
|
const { filename, body } = await deps2.api.downloadJob(creds, jobId, download);
|
|
2037
2296
|
const path = opts.output ?? filename;
|
|
2038
2297
|
await writeFile(path, body);
|
|
@@ -2042,7 +2301,7 @@ function registerDownloadCommand(program, deps2) {
|
|
|
2042
2301
|
}
|
|
2043
2302
|
|
|
2044
2303
|
// src/commands/jobs.ts
|
|
2045
|
-
import
|
|
2304
|
+
import kleur6 from "kleur";
|
|
2046
2305
|
|
|
2047
2306
|
// src/flags.ts
|
|
2048
2307
|
function parseIntegerFlag(value, flag, range = {}) {
|
|
@@ -2056,7 +2315,7 @@ function parseIntegerFlag(value, flag, range = {}) {
|
|
|
2056
2315
|
}
|
|
2057
2316
|
|
|
2058
2317
|
// src/watch.ts
|
|
2059
|
-
import
|
|
2318
|
+
import kleur5 from "kleur";
|
|
2060
2319
|
var STOP = /* @__PURE__ */ new Set(["completed", "failed", "cancelled", "paused"]);
|
|
2061
2320
|
function isStop(status) {
|
|
2062
2321
|
return typeof status === "string" && STOP.has(status);
|
|
@@ -2079,15 +2338,15 @@ function render(out, frame) {
|
|
|
2079
2338
|
}
|
|
2080
2339
|
if (frame.type === "status") {
|
|
2081
2340
|
out.progress.done();
|
|
2082
|
-
out.info(`${
|
|
2341
|
+
out.info(`${kleur5.gray("\u2022")} ${String(frame.status)}`);
|
|
2083
2342
|
} else if (frame.type === "error") {
|
|
2084
2343
|
out.progress.done();
|
|
2085
|
-
out.info(`${
|
|
2344
|
+
out.info(`${kleur5.red("error:")} ${String(frame.message ?? "job failed")}`);
|
|
2086
2345
|
}
|
|
2087
2346
|
}
|
|
2088
2347
|
async function watchJob(api, out, creds, jobId) {
|
|
2089
2348
|
out.progress.done();
|
|
2090
|
-
if (!out.json) out.info(`${
|
|
2349
|
+
if (!out.json) out.info(`${kleur5.cyan("\u2192")} Streaming progress for ${jobId} \u2026`);
|
|
2091
2350
|
for await (const frame of api.streamJob(creds, jobId)) {
|
|
2092
2351
|
if (frame.type === "heartbeat") continue;
|
|
2093
2352
|
const status = stopStatusOf(frame);
|
|
@@ -2108,8 +2367,8 @@ function finish(out, status) {
|
|
|
2108
2367
|
out.progress.done();
|
|
2109
2368
|
if (!out.json) {
|
|
2110
2369
|
if (status === "completed") out.success("completed");
|
|
2111
|
-
else if (status === "paused") out.info(
|
|
2112
|
-
else out.info(
|
|
2370
|
+
else if (status === "paused") out.info(kleur5.yellow("paused \u2014 resume it, then watch again"));
|
|
2371
|
+
else out.info(kleur5.red(`\u2717 ${status}`));
|
|
2113
2372
|
}
|
|
2114
2373
|
return status;
|
|
2115
2374
|
}
|
|
@@ -2131,7 +2390,7 @@ function watchOutcomeError(results) {
|
|
|
2131
2390
|
}
|
|
2132
2391
|
function listLine(job) {
|
|
2133
2392
|
const matched = `${String(job.matchedRows).padStart(6)}/${String(job.totalRows).padStart(6)} matched`;
|
|
2134
|
-
return `${job.id} ${job.status.padEnd(10)} ${matched} ${
|
|
2393
|
+
return `${job.id} ${job.status.padEnd(10)} ${matched} ${kleur6.gray(job.name ?? "")}`;
|
|
2135
2394
|
}
|
|
2136
2395
|
var JOB_ACTIONS = {
|
|
2137
2396
|
pause: { description: "Pause a running job (the worker stops between match queries)", past: "paused" },
|
|
@@ -2154,7 +2413,7 @@ function registerJobCommands(program, deps2) {
|
|
|
2154
2413
|
const creds = await requireCredentials(deps2.store, deps2.env);
|
|
2155
2414
|
const jobs = await deps2.api.listJobs(creds, { limit, ...status ? { status: status.data } : {} });
|
|
2156
2415
|
if (out.json) return out.data(jobs);
|
|
2157
|
-
if (jobs.length === 0) return out.info(
|
|
2416
|
+
if (jobs.length === 0) return out.info(kleur6.gray("No jobs."));
|
|
2158
2417
|
for (const job of jobs) out.info(listLine(job));
|
|
2159
2418
|
});
|
|
2160
2419
|
program.command("status <jobId>").description("Print a job snapshot as JSON (counts, status, timestamps)").option("--json", "Accepted for symmetry: the output is always JSON", false).action(async (jobId) => {
|
|
@@ -2182,7 +2441,7 @@ function registerJobCommands(program, deps2) {
|
|
|
2182
2441
|
}
|
|
2183
2442
|
|
|
2184
2443
|
// src/commands/skills.ts
|
|
2185
|
-
import
|
|
2444
|
+
import kleur7 from "kleur";
|
|
2186
2445
|
|
|
2187
2446
|
// src/skills/manage.ts
|
|
2188
2447
|
import { promises as fs3 } from "fs";
|
|
@@ -2201,11 +2460,31 @@ async function walk(dir, root = dir) {
|
|
|
2201
2460
|
}
|
|
2202
2461
|
return found.sort();
|
|
2203
2462
|
}
|
|
2463
|
+
function table(rows, keyOf) {
|
|
2464
|
+
const lines = Object.entries(rows).map(([key, description]) => `| \`${keyOf(key)}\` | ${description} |`);
|
|
2465
|
+
return ["| Column | What it holds |", "| --- | --- |", ...lines].join("\n");
|
|
2466
|
+
}
|
|
2467
|
+
function exportColumnsMarkdown() {
|
|
2468
|
+
return [
|
|
2469
|
+
"## Every column",
|
|
2470
|
+
"",
|
|
2471
|
+
table(RESULT_COLUMN_DESCRIPTIONS, (key) => key),
|
|
2472
|
+
"",
|
|
2473
|
+
"## Fields `--wcvp-extra` can append",
|
|
2474
|
+
"",
|
|
2475
|
+
"Columns of the accepted taxon's WCVP record, exported as `wcvp_<field>`, e.g. `--wcvp-extra ipni_id,powo_id`.",
|
|
2476
|
+
"",
|
|
2477
|
+
table(WCVP_EXTRA_DESCRIPTIONS, (key) => key)
|
|
2478
|
+
].join("\n");
|
|
2479
|
+
}
|
|
2204
2480
|
async function renderSkill(sourceDir, version2) {
|
|
2205
2481
|
const files = /* @__PURE__ */ new Map();
|
|
2206
2482
|
for (const path of await walk(sourceDir)) {
|
|
2207
2483
|
const text = await fs2.readFile(join3(sourceDir, path), "utf8");
|
|
2208
|
-
files.set(
|
|
2484
|
+
files.set(
|
|
2485
|
+
path,
|
|
2486
|
+
text.replaceAll("{{version}}", version2).replaceAll("{{exportColumns}}", exportColumnsMarkdown())
|
|
2487
|
+
);
|
|
2209
2488
|
}
|
|
2210
2489
|
return files;
|
|
2211
2490
|
}
|
|
@@ -2435,7 +2714,7 @@ function tilde(path, home) {
|
|
|
2435
2714
|
}
|
|
2436
2715
|
function printEntries(out, result, home) {
|
|
2437
2716
|
for (const { dir, path, action } of result.entries) {
|
|
2438
|
-
const line = `${SERVES[dir]}: ${ACTION_TEXT[action]} ${
|
|
2717
|
+
const line = `${SERVES[dir]}: ${ACTION_TEXT[action]} ${kleur7.gray(tilde(path, home))}`;
|
|
2439
2718
|
if (action.startsWith("skipped")) out.warn(line);
|
|
2440
2719
|
else out.success(line);
|
|
2441
2720
|
}
|
|
@@ -2455,17 +2734,17 @@ function registerSkillsCommands(program, deps2, version2) {
|
|
|
2455
2734
|
const result = await installSkill(paths(), deps2.skillSource, version2, agents, { force: opts.force });
|
|
2456
2735
|
if (out.json) return out.data({ version: version2, agents, ...result });
|
|
2457
2736
|
out.info(
|
|
2458
|
-
`${
|
|
2737
|
+
`${kleur7.bold("planttaxomatcher")} skill ${version2} \u2192 ${kleur7.gray(tilde(result.canonical, deps2.home))}`
|
|
2459
2738
|
);
|
|
2460
2739
|
printEntries(out, result, deps2.home);
|
|
2461
|
-
for (const agent of agents) if (RELOAD_HINTS[agent]) out.info(
|
|
2462
|
-
out.info(
|
|
2740
|
+
for (const agent of agents) if (RELOAD_HINTS[agent]) out.info(kleur7.gray(RELOAD_HINTS[agent]));
|
|
2741
|
+
out.info(kleur7.gray(`Try it: ask your agent to "match the plant names in my CSV with PlantTaxoMatcher".`));
|
|
2463
2742
|
});
|
|
2464
2743
|
skills.command("uninstall").description("Remove the links and copies this CLI installed, then its own copy of the skill").option("--force", "Also remove copies edited by hand", false).option("--json", "Print the result as JSON", false).action(async (opts) => {
|
|
2465
2744
|
const out = createOutput(deps2.stdout, deps2.stderr, opts.json, deps2.now);
|
|
2466
2745
|
const result = await uninstallSkill(paths(), { force: opts.force });
|
|
2467
2746
|
if (out.json) return out.data(result);
|
|
2468
|
-
if (result.entries.length === 0) out.info(
|
|
2747
|
+
if (result.entries.length === 0) out.info(kleur7.gray("No installed skill links found."));
|
|
2469
2748
|
printEntries(out, result, deps2.home);
|
|
2470
2749
|
});
|
|
2471
2750
|
skills.command("status").description("Show where the skill is installed and which copy each agent loads").option("--json", "Print the status as JSON", false).action(async (opts) => {
|
|
@@ -2474,28 +2753,28 @@ function registerSkillsCommands(program, deps2, version2) {
|
|
|
2474
2753
|
if (out.json) return out.data({ cliVersion: version2, ...status });
|
|
2475
2754
|
const { canonical } = status;
|
|
2476
2755
|
out.info(
|
|
2477
|
-
canonical.version ? `Skill ${canonical.version}${canonical.pristine ? "" : " (edited)"} at ${
|
|
2756
|
+
canonical.version ? `Skill ${canonical.version}${canonical.pristine ? "" : " (edited)"} at ${kleur7.gray(tilde(canonical.path, deps2.home))}` : kleur7.gray("Not installed. Run `planttaxomatcher skills install`.")
|
|
2478
2757
|
);
|
|
2479
2758
|
for (const entry of status.entries) {
|
|
2480
2759
|
if (entry.kind !== "missing")
|
|
2481
|
-
out.info(` ${entry.kind.padEnd(8)} ${
|
|
2760
|
+
out.info(` ${entry.kind.padEnd(8)} ${kleur7.gray(tilde(entry.path, deps2.home))}`);
|
|
2482
2761
|
}
|
|
2483
2762
|
for (const { agent, installed, loads, state } of status.agents) {
|
|
2484
2763
|
const entry = status.entries.find((e) => e.dir === loads);
|
|
2485
2764
|
const where = entry ? `${state} in ${tilde(entry.path, deps2.home)}` : "no skill";
|
|
2486
|
-
out.info(` ${AGENT_NAMES[agent].padEnd(12)} ${installed ? where :
|
|
2765
|
+
out.info(` ${AGENT_NAMES[agent].padEnd(12)} ${installed ? where : kleur7.gray(`not found (${where})`)}`);
|
|
2487
2766
|
}
|
|
2488
2767
|
});
|
|
2489
2768
|
}
|
|
2490
2769
|
|
|
2491
2770
|
// src/commands/submit.ts
|
|
2492
2771
|
import { parse as parsePath } from "path";
|
|
2493
|
-
import
|
|
2772
|
+
import kleur9 from "kleur";
|
|
2494
2773
|
|
|
2495
2774
|
// src/dry-run.ts
|
|
2496
2775
|
import { promises as fs4, createReadStream } from "fs";
|
|
2497
2776
|
import Papa from "papaparse";
|
|
2498
|
-
import
|
|
2777
|
+
import kleur8 from "kleur";
|
|
2499
2778
|
async function readSample(file, sampleLimit) {
|
|
2500
2779
|
const lower = file.toLowerCase();
|
|
2501
2780
|
if (lower.endsWith(".json")) {
|
|
@@ -2600,23 +2879,23 @@ async function buildDryRunReport(file, opts) {
|
|
|
2600
2879
|
}
|
|
2601
2880
|
function printDryRunReport(report, log) {
|
|
2602
2881
|
log("");
|
|
2603
|
-
log(
|
|
2882
|
+
log(kleur8.bold("Dry-run preview"));
|
|
2604
2883
|
log(
|
|
2605
|
-
` Read ${
|
|
2884
|
+
` Read ${kleur8.cyan(report.rowsRead)} rows \xB7 ${kleur8.yellow(report.nullNameCount)} with no usable name \xB7 ${kleur8.yellow(report.qualifierFlagCount)} with qualifier flags`
|
|
2606
2885
|
);
|
|
2607
2886
|
if (report.idTypeDetection) {
|
|
2608
2887
|
const d = report.idTypeDetection;
|
|
2609
2888
|
const pct = Math.round(d.dominantConfidence * 100);
|
|
2610
|
-
log(` ID-column detection: ${
|
|
2889
|
+
log(` ID-column detection: ${kleur8.green(d.dominant ?? "\u2014")} (${pct}% of ${d.sampleSize} sampled)`);
|
|
2611
2890
|
if (d.minorityExamples.length > 0) {
|
|
2612
2891
|
log(` Minority examples:`);
|
|
2613
2892
|
for (const m of d.minorityExamples) {
|
|
2614
|
-
log(` row ${m.rowIndex + 1}: ${
|
|
2893
|
+
log(` row ${m.rowIndex + 1}: ${kleur8.gray(m.value)} \u2192 ${kleur8.dim(m.type)}`);
|
|
2615
2894
|
}
|
|
2616
2895
|
}
|
|
2617
2896
|
}
|
|
2618
2897
|
log("");
|
|
2619
|
-
log(
|
|
2898
|
+
log(kleur8.bold("First rows after normalization:"));
|
|
2620
2899
|
log(` ${"#".padStart(4)} ${"input".padEnd(36)} ${"normalized".padEnd(36)} flags`);
|
|
2621
2900
|
for (const r of report.sampleRows) {
|
|
2622
2901
|
const idx = String(r.rowIndex + 1).padStart(4);
|
|
@@ -2677,7 +2956,9 @@ function buildJobConfig(opts) {
|
|
|
2677
2956
|
speciesLevelAcceptedOnly: opts.speciesLevel,
|
|
2678
2957
|
acceptUnconfirmedAuthor: opts.ignoreAuthor,
|
|
2679
2958
|
exportConfirmedOnly: false,
|
|
2680
|
-
|
|
2959
|
+
referential: parseBackbone(opts.backbone),
|
|
2960
|
+
...opts.referential ? { referentialVersion: opts.referential } : {},
|
|
2961
|
+
...opts.area?.trim() ? { area: opts.area.trim() } : {}
|
|
2681
2962
|
};
|
|
2682
2963
|
const checked = jobConfigSchema.safeParse(config);
|
|
2683
2964
|
if (!checked.success) {
|
|
@@ -2700,13 +2981,20 @@ function summarize(submitted, finals, creds) {
|
|
|
2700
2981
|
url: jobUrl(creds, job.id)
|
|
2701
2982
|
}));
|
|
2702
2983
|
}
|
|
2984
|
+
async function checkAgainstServer(deps2, creds, config) {
|
|
2985
|
+
const backbone = config.referential ?? "wcvp";
|
|
2986
|
+
if (config.referentialVersion) {
|
|
2987
|
+
requireVersion(await fetchVersions(deps2, creds, backbone), backbone, config.referentialVersion);
|
|
2988
|
+
}
|
|
2989
|
+
if (config.area) config.area = requireArea(await deps2.api.listAreas(creds), config.area).code;
|
|
2990
|
+
}
|
|
2703
2991
|
function jobNameFor(file, fileCount, name) {
|
|
2704
2992
|
return fileCount === 1 && name?.trim() ? name.trim() : parsePath(file).name;
|
|
2705
2993
|
}
|
|
2706
2994
|
async function confirmDryRun(deps2, out, files, opts) {
|
|
2707
2995
|
const previewRows = Number(opts.dryRunRows);
|
|
2708
2996
|
for (const file of files) {
|
|
2709
|
-
if (files.length > 1) out.info(
|
|
2997
|
+
if (files.length > 1) out.info(kleur9.bold(`
|
|
2710
2998
|
${file}`));
|
|
2711
2999
|
const report = await buildDryRunReport(file, {
|
|
2712
3000
|
nameColumn: opts.nameColumn,
|
|
@@ -2741,13 +3029,17 @@ function registerSubmitCommand(program, deps2) {
|
|
|
2741
3029
|
"--ignore-author",
|
|
2742
3030
|
"Accept a unique canonical name match whose author couldn't be confirmed instead of sending it to review (a CONFLICTING author still reviews)",
|
|
2743
3031
|
false
|
|
2744
|
-
).option("--keep-infraspecific", "Deprecated \u2014 now the default; has no effect", false).option("--allow-llm", "Allow the Layer 7 LLM to break ties between competing matches", false).option("--llm-cap-cents <cents>", "Max LLM spend in cents", "500").option(
|
|
3032
|
+
).option("--keep-infraspecific", "Deprecated \u2014 now the default; has no effect", false).option("--allow-llm", "Allow the Layer 7 LLM to break ties between competing matches", false).option("--llm-cap-cents <cents>", "Max LLM spend in cents", "500").option("--backbone <name>", "Backbone to match against: wcvp or wfo", "wcvp").option(
|
|
2745
3033
|
"--referential <version>",
|
|
2746
|
-
"
|
|
3034
|
+
"Backbone version to match against (default: the one `planttaxomatcher versions` marks as default)"
|
|
3035
|
+
).option(
|
|
3036
|
+
"--area <code>",
|
|
3037
|
+
"WGSRPD level-3 area (botanical country, e.g. FRA): an ambiguous match is narrowed to the taxa native there. See `planttaxomatcher areas`"
|
|
2747
3038
|
).option("--no-watch", "Return as soon as the jobs are created").option("--dry-run", "Preview the first rows locally, then confirm before uploading", false).option("--dry-run-rows <n>", "Rows to show in the dry-run preview", "10").option("-y, --yes", "Skip the --dry-run confirmation (needed without a terminal)", false).option("--json", "Print the created jobs as a JSON array (with --watch: once they end)", false).action(async (patterns, opts) => {
|
|
2748
3039
|
const out = createOutput(deps2.stdout, deps2.stderr, opts.json, deps2.now);
|
|
2749
3040
|
const config = buildJobConfig(opts);
|
|
2750
3041
|
const creds = await requireCredentials(deps2.store, deps2.env);
|
|
3042
|
+
await checkAgainstServer(deps2, creds, config);
|
|
2751
3043
|
const files = await expandInputs(patterns);
|
|
2752
3044
|
if (opts.name && files.length > 1)
|
|
2753
3045
|
out.warn("--name ignored for a multi-file submit; each job is named after its file");
|
|
@@ -2757,11 +3049,11 @@ function registerSubmitCommand(program, deps2) {
|
|
|
2757
3049
|
);
|
|
2758
3050
|
}
|
|
2759
3051
|
if (files.length > 1) {
|
|
2760
|
-
out.info(`${
|
|
2761
|
-
for (const file of files) out.info(
|
|
3052
|
+
out.info(`${kleur9.cyan("\u2192")} ${files.length} files matched:`);
|
|
3053
|
+
for (const file of files) out.info(kleur9.gray(` ${file}`));
|
|
2762
3054
|
}
|
|
2763
3055
|
if (opts.dryRun && !await confirmDryRun(deps2, out, files, opts)) {
|
|
2764
|
-
out.info(
|
|
3056
|
+
out.info(kleur9.gray("Aborted."));
|
|
2765
3057
|
return;
|
|
2766
3058
|
}
|
|
2767
3059
|
const submitted = [];
|
|
@@ -2769,16 +3061,16 @@ function registerSubmitCommand(program, deps2) {
|
|
|
2769
3061
|
try {
|
|
2770
3062
|
for (const file of files) {
|
|
2771
3063
|
const name = jobNameFor(file, files.length, opts.name);
|
|
2772
|
-
out.info(`${
|
|
3064
|
+
out.info(`${kleur9.cyan("\u2192")} Uploading ${file} \u2026`);
|
|
2773
3065
|
const job = await deps2.api.submitJob(creds, file, config, name || null);
|
|
2774
|
-
out.success(`Job created: ${job.id} ${
|
|
3066
|
+
out.success(`Job created: ${job.id} ${kleur9.gray(name)}`);
|
|
2775
3067
|
out.info(` rows=${job.totalRows} uniqueQueries=${job.uniqueQueries}`);
|
|
2776
3068
|
submitted.push({ file, job });
|
|
2777
3069
|
}
|
|
2778
3070
|
if (opts.watch) {
|
|
2779
3071
|
const watchOut = opts.json ? createOutput(deps2.stderr, deps2.stderr, false, deps2.now) : out;
|
|
2780
3072
|
for (const { file, job } of submitted) {
|
|
2781
|
-
if (submitted.length > 1) watchOut.info(
|
|
3073
|
+
if (submitted.length > 1) watchOut.info(kleur9.bold(`
|
|
2782
3074
|
[${file}] ${job.id}`));
|
|
2783
3075
|
finals.push({ id: job.id, status: await watchJob(deps2.api, watchOut, creds, job.id) });
|
|
2784
3076
|
}
|
|
@@ -2813,6 +3105,7 @@ function buildProgram(deps2, version2) {
|
|
|
2813
3105
|
registerAuthCommands(program, deps2);
|
|
2814
3106
|
registerJobCommands(program, deps2);
|
|
2815
3107
|
registerSubmitCommand(program, deps2);
|
|
3108
|
+
registerCatalogCommands(program, deps2);
|
|
2816
3109
|
registerDownloadCommand(program, deps2);
|
|
2817
3110
|
registerSkillsCommands(program, deps2, version2);
|
|
2818
3111
|
return program;
|
|
@@ -2823,9 +3116,9 @@ async function runCli(argv, deps2, version2) {
|
|
|
2823
3116
|
return 0;
|
|
2824
3117
|
} catch (err) {
|
|
2825
3118
|
const { exitCode, message, hint } = describeError(err);
|
|
2826
|
-
if (message) deps2.stderr.write(`${
|
|
3119
|
+
if (message) deps2.stderr.write(`${kleur10.red("error:")} ${message}
|
|
2827
3120
|
`);
|
|
2828
|
-
if (hint) deps2.stderr.write(`${
|
|
3121
|
+
if (hint) deps2.stderr.write(`${kleur10.gray(hint)}
|
|
2829
3122
|
`);
|
|
2830
3123
|
return exitCode;
|
|
2831
3124
|
}
|
package/package.json
CHANGED
|
@@ -62,9 +62,21 @@ row of an XLSX sheet) and choose the columns:
|
|
|
62
62
|
- `--family-column`, `--genus-column`, `--rank-column` when present: they help break ties.
|
|
63
63
|
- `--id-column` when rows already carry an identifier, with `--id-type wcvp` or `--id-type gbif` if you know which.
|
|
64
64
|
|
|
65
|
+
Then the reference to match against:
|
|
66
|
+
|
|
67
|
+
- **Backbone**: WCVP by default. Pass `--backbone wfo` only when the user asks for World Flora Online.
|
|
68
|
+
- **Version**: leave `--referential` out to use the server's default.
|
|
69
|
+
`planttaxomatcher versions --json` lists the versions; the one with
|
|
70
|
+
`"isDefault": true` is used when none is named. Pass `--referential <version>`
|
|
71
|
+
only when the user names one, or needs a team-edited version.
|
|
72
|
+
- **Area**: when the user says where the plants grow (a country, a regional
|
|
73
|
+
flora), pass `--area <code>` so an ambiguous name is narrowed to the taxa
|
|
74
|
+
native there. Find the WGSRPD level-3 code with
|
|
75
|
+
`planttaxomatcher areas --search <country> --json`, e.g. `FRA` for France.
|
|
76
|
+
|
|
65
77
|
Ask the user when the name column is not obvious. Keep the defaults for
|
|
66
78
|
everything else unless the user asks; `references/submit-options.md` lists
|
|
67
|
-
every option (author handling, species-level roll-up,
|
|
79
|
+
every option (author handling, species-level roll-up, row filter, LLM).
|
|
68
80
|
|
|
69
81
|
## 3. Submit
|
|
70
82
|
|
|
@@ -93,11 +105,29 @@ planttaxomatcher download <jobId> --format csv --output results.csv --json
|
|
|
93
105
|
```
|
|
94
106
|
|
|
95
107
|
The export keeps every column of the user's file and adds `planttaxomatcher_*`
|
|
96
|
-
and `wcvp_*` columns
|
|
97
|
-
|
|
98
|
-
|
|
99
|
-
|
|
100
|
-
|
|
108
|
+
and `wcvp_*` columns. The cleaned name to hand back is
|
|
109
|
+
**`wcvp_accepted_name_with_author`**: the accepted name with its author, ready
|
|
110
|
+
to cite (e.g. *Calicotome spinosa (L.) Link*), filled even when the input was a
|
|
111
|
+
synonym. `wcvp_accepted_name` is the same without the author, and
|
|
112
|
+
`wcvp_matched_name` is what the input matched before synonyms were resolved.
|
|
113
|
+
|
|
114
|
+
The filters of the web app's download dialog. Check what each keeps first
|
|
115
|
+
with `planttaxomatcher download <jobId> --counts --json`:
|
|
116
|
+
|
|
117
|
+
- `--confirmed-only`: only rows accepted automatically (grade A) or by a reviewer.
|
|
118
|
+
- `--resolved-only`: drop the rows with no accepted name (no match, error, rejected).
|
|
119
|
+
- `--dedupe`: one row per accepted taxon (synonyms and duplicates collapse).
|
|
120
|
+
- `--taxon-level species_below` (drop genus and family matches) or `--taxon-level species_only`.
|
|
121
|
+
|
|
122
|
+
To shape the file:
|
|
123
|
+
|
|
124
|
+
- `--columns` keeps only the listed columns. Start with the user's own name column (the one passed to `--name-column`), e.g. `--columns <name column>,wcvp_accepted_name_with_author,planttaxomatcher_grade,planttaxomatcher_review_status`.
|
|
125
|
+
- `--wcvp-extra ipni_id,powo_id` appends WCVP fields as `wcvp_<field>` columns.
|
|
126
|
+
- `--format xlsx`, or `--bundle` for a ZIP with a NOTICE.md that cites the backbone version.
|
|
127
|
+
|
|
128
|
+
`planttaxomatcher download <jobId> --list-columns --json` lists every column
|
|
129
|
+
with what it holds; `references/results.md` describes them all, with the grades
|
|
130
|
+
and review states.
|
|
101
131
|
|
|
102
132
|
## 6. Report back
|
|
103
133
|
|
|
@@ -105,7 +135,9 @@ Summarise from the status counts and the export:
|
|
|
105
135
|
|
|
106
136
|
- rows matched and accepted automatically (grade A),
|
|
107
137
|
- rows waiting for review (grade B or C, ambiguous),
|
|
108
|
-
- rows with no match or an error, with a few examples
|
|
138
|
+
- rows with no match or an error, with a few examples,
|
|
139
|
+
|
|
140
|
+
quoting names as input → `wcvp_accepted_name_with_author`.
|
|
109
141
|
|
|
110
142
|
Give the job's `url`: reviewing happens in the web app, not the CLI. Call a
|
|
111
143
|
name accepted only when `planttaxomatcher_review_status` is `not_required`,
|
|
@@ -114,4 +146,5 @@ name accepted only when `planttaxomatcher_review_status` is `not_required`,
|
|
|
114
146
|
## Other commands
|
|
115
147
|
|
|
116
148
|
- `planttaxomatcher list --json`: recent jobs (`--status completed`, `--limit 50`).
|
|
149
|
+
- `planttaxomatcher versions --json` and `planttaxomatcher areas --json`: what `--referential` and `--area` accept.
|
|
117
150
|
- `planttaxomatcher pause <jobId>`, `planttaxomatcher resume <jobId>`, `planttaxomatcher cancel <jobId>`: control a running job; each takes `--json`. Cancelling keeps the rows already matched. Ask before cancelling.
|
|
@@ -2,37 +2,20 @@
|
|
|
2
2
|
|
|
3
3
|
Every row of the user's file comes back with its original columns, plus the
|
|
4
4
|
columns below. `planttaxomatcher download <jobId> --list-columns --json` lists
|
|
5
|
-
them
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
| `planttaxomatcher_input_normalized` | The name as matched, after cleaning |
|
|
20
|
-
| `planttaxomatcher_input_qualifier` | A qualifier found in the input (`cf`, `aff`, `sp`, `aggregate`…) |
|
|
21
|
-
|
|
22
|
-
## WCVP columns
|
|
23
|
-
|
|
24
|
-
| Column | Meaning |
|
|
25
|
-
| ------------------------------------------------------------------------------ | -------------------------------------------------- |
|
|
26
|
-
| `wcvp_matched_name` | The WCVP name the input matched (may be a synonym) |
|
|
27
|
-
| `wcvp_matched_taxonomic_status` | Its status in WCVP: `Accepted`, `Synonym`, … |
|
|
28
|
-
| `wcvp_accepted_name`, `wcvp_accepted_author`, `wcvp_accepted_name_with_author` | The accepted name the match resolves to |
|
|
29
|
-
| `wcvp_accepted_taxon_id` | WCVP id of the accepted taxon |
|
|
30
|
-
| `wcvp_family`, `wcvp_rank` | Family and rank of the accepted taxon |
|
|
31
|
-
| `ipni_lsid`, `powo_url` | Links to IPNI and Plants of the World Online |
|
|
32
|
-
|
|
33
|
-
`--wcvp-extra` appends more WCVP fields as `wcvp_<key>`, for example
|
|
34
|
-
`ipni_id`, `powo_id`, `geographic_area`, `lifeform_description`,
|
|
35
|
-
`first_published`.
|
|
5
|
+
them for a given job (the Pl@ntNet columns only exist when the server has
|
|
6
|
+
Pl@ntNet enabled).
|
|
7
|
+
|
|
8
|
+
## The columns that matter most
|
|
9
|
+
|
|
10
|
+
- `wcvp_accepted_name_with_author`: the cleaned name to hand back, the accepted
|
|
11
|
+
name with its author (e.g. *Calicotome spinosa (L.) Link*), even when the
|
|
12
|
+
input was a synonym.
|
|
13
|
+
- `planttaxomatcher_grade` and `planttaxomatcher_review_status`: how far to
|
|
14
|
+
trust it (below).
|
|
15
|
+
- `planttaxomatcher_flags` and `planttaxomatcher_reason`: why a row needs a look.
|
|
16
|
+
- `wcvp_accepted_taxon_id`, `ipni_lsid`, `powo_url`: identifiers and links for
|
|
17
|
+
the accepted taxon (WCVP jobs; a WFO job's id is in
|
|
18
|
+
`planttaxomatcher_accepted_identifier`).
|
|
36
19
|
|
|
37
20
|
## Grades
|
|
38
21
|
|
|
@@ -53,3 +36,5 @@ them all for a given job.
|
|
|
53
36
|
|
|
54
37
|
Only `not_required`, `accepted` and `overridden` rows carry a name the user
|
|
55
38
|
can rely on. `--confirmed-only` exports just those.
|
|
39
|
+
|
|
40
|
+
{{exportColumns}}
|
|
@@ -16,6 +16,16 @@ something else; they are what the web app uses.
|
|
|
16
16
|
| `--id-type <type>` | What `--id-column` holds: `auto` (default), `wcvp`, `gbif` |
|
|
17
17
|
| `--filter-column <name>` and `--filter-value <value>` | Only match rows whose column equals the value (trimmed, case-insensitive), e.g. `--filter-column kingdom --filter-value Plantae` |
|
|
18
18
|
|
|
19
|
+
## Backbone, version and area
|
|
20
|
+
|
|
21
|
+
| Option | Default | Effect |
|
|
22
|
+
| --- | --- | --- |
|
|
23
|
+
| `--backbone <name>` | `wcvp` | `wcvp` (World Checklist of Vascular Plants) or `wfo` (World Flora Online) |
|
|
24
|
+
| `--referential <version>` | the server's default | Backbone version, e.g. `wcvp-v14`. `planttaxomatcher versions --json` lists them and marks the default (`isDefault`); a team-edited version is listed with `kind: derived` |
|
|
25
|
+
| `--area <code>` | none | WGSRPD level-3 area (botanical country), e.g. `FRA`: an ambiguous name is narrowed to the taxa native there. `planttaxomatcher areas --search <text> --json` finds the code |
|
|
26
|
+
|
|
27
|
+
The CLI checks `--referential` and `--area` against the server before uploading, and names the valid values on a typo.
|
|
28
|
+
|
|
19
29
|
## Matching
|
|
20
30
|
|
|
21
31
|
| Option | Default | Effect |
|
|
@@ -23,7 +33,6 @@ something else; they are what the web app uses.
|
|
|
23
33
|
| `--author-mode <mode>` | `prefer` | `ignore`, `prefer` or `strict` handling of authorship |
|
|
24
34
|
| `--ignore-author` | off | Accept a unique name match whose author could not be confirmed, instead of sending it to review. A conflicting author still goes to review |
|
|
25
35
|
| `--species-level` | off | Roll varieties, subspecies and forms up to their accepted species |
|
|
26
|
-
| `--referential <version>` | team default | WCVP snapshot to match against, e.g. `wcvp-v14` |
|
|
27
36
|
| `--allow-llm` | off | Let a language model break ties between competing matches |
|
|
28
37
|
| `--llm-cap-cents <cents>` | `500` | Spending cap for `--allow-llm` |
|
|
29
38
|
| `--review-mode <mode>` | `recommended` | `off`, `recommended` or `strict` |
|