@plantnet/planttaxomatcher 0.1.0 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +87 -64
- package/package.json +43 -41
package/dist/index.js
CHANGED
|
@@ -101,7 +101,16 @@ var jobConfigSchema = z3.object({
|
|
|
101
101
|
* resolved version on `jobs.referential_version` so re-runs are
|
|
102
102
|
* reproducible even if a newer snapshot lands later.
|
|
103
103
|
*/
|
|
104
|
-
referentialVersion: z3.string().min(1).optional()
|
|
104
|
+
referentialVersion: z3.string().min(1).optional(),
|
|
105
|
+
/**
|
|
106
|
+
* Row filter: keep only rows whose `filterColumn` value equals
|
|
107
|
+
* `filterValue` (trimmed, case-insensitive); all other rows are skipped at
|
|
108
|
+
* ingest (no match query created). Both must be set for the filter to
|
|
109
|
+
* apply. Lets a user match a subset of a mixed file (e.g. only
|
|
110
|
+
* `kingdom = Plantae`).
|
|
111
|
+
*/
|
|
112
|
+
filterColumn: z3.string().nullable().optional(),
|
|
113
|
+
filterValue: z3.string().nullable().optional()
|
|
105
114
|
});
|
|
106
115
|
var wcvpSnapshotSummarySchema = z3.object({
|
|
107
116
|
version: z3.string(),
|
|
@@ -785,6 +794,7 @@ var apiClient = {
|
|
|
785
794
|
const params = new URLSearchParams({ format: opts.format });
|
|
786
795
|
if (opts.confirmedOnly) params.set("confirmedOnly", "true");
|
|
787
796
|
if (opts.bundle) params.set("bundle", "true");
|
|
797
|
+
if (opts.delimiter) params.set("delimiter", opts.delimiter);
|
|
788
798
|
if (opts.columns) params.set("columns", opts.columns);
|
|
789
799
|
if (opts.wcvpExtra) params.set("wcvpExtra", opts.wcvpExtra);
|
|
790
800
|
const url = new URL(`/v1/jobs/${id}/download?${params}`, creds.server).toString();
|
|
@@ -795,7 +805,7 @@ var apiClient = {
|
|
|
795
805
|
}
|
|
796
806
|
const disp = res.headers.get("content-disposition") ?? "";
|
|
797
807
|
const m = /filename="?([^"]+)"?/.exec(disp);
|
|
798
|
-
const ext = opts.bundle ? "zip" : opts.format === "csv" ? "csv" : opts.format === "json" ? "json" : "ndjson";
|
|
808
|
+
const ext = opts.bundle ? "zip" : opts.format === "csv" ? "csv" : opts.format === "json" ? "json" : opts.format === "xlsx" ? "xlsx" : "ndjson";
|
|
799
809
|
const filename = m?.[1] ?? `planttaxomatcher_${id}.${ext}`;
|
|
800
810
|
const buf = Buffer.from(await res.arrayBuffer());
|
|
801
811
|
return { filename, body: buf };
|
|
@@ -810,20 +820,25 @@ var apiClient = {
|
|
|
810
820
|
const reader = res.body.getReader();
|
|
811
821
|
const decoder = new TextDecoder("utf-8");
|
|
812
822
|
let buf = "";
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
818
|
-
|
|
819
|
-
|
|
820
|
-
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
|
|
823
|
+
try {
|
|
824
|
+
for (; ; ) {
|
|
825
|
+
const { done, value } = await reader.read();
|
|
826
|
+
if (done) break;
|
|
827
|
+
buf += decoder.decode(value, { stream: true });
|
|
828
|
+
let nl;
|
|
829
|
+
while ((nl = buf.indexOf("\n")) >= 0) {
|
|
830
|
+
const line = buf.slice(0, nl).trim();
|
|
831
|
+
buf = buf.slice(nl + 1);
|
|
832
|
+
if (!line) continue;
|
|
833
|
+
try {
|
|
834
|
+
yield JSON.parse(line);
|
|
835
|
+
} catch {
|
|
836
|
+
}
|
|
825
837
|
}
|
|
826
838
|
}
|
|
839
|
+
} finally {
|
|
840
|
+
await reader.cancel().catch(() => {
|
|
841
|
+
});
|
|
827
842
|
}
|
|
828
843
|
}
|
|
829
844
|
};
|
|
@@ -863,9 +878,8 @@ async function requireCredentials() {
|
|
|
863
878
|
}
|
|
864
879
|
|
|
865
880
|
// src/dry-run.ts
|
|
866
|
-
import { promises as fs3 } from "fs";
|
|
867
|
-
import
|
|
868
|
-
import { parse as parseCsvStream } from "csv-parse";
|
|
881
|
+
import { promises as fs3, createReadStream } from "fs";
|
|
882
|
+
import Papa from "papaparse";
|
|
869
883
|
import kleur from "kleur";
|
|
870
884
|
async function readSample(file, sampleLimit) {
|
|
871
885
|
const lower = file.toLowerCase();
|
|
@@ -873,54 +887,47 @@ async function readSample(file, sampleLimit) {
|
|
|
873
887
|
const text = await fs3.readFile(file, "utf8");
|
|
874
888
|
const arr = JSON.parse(text);
|
|
875
889
|
if (!Array.isArray(arr)) throw new Error("JSON file must be an array of row objects");
|
|
876
|
-
const
|
|
890
|
+
const out = [];
|
|
877
891
|
for (const v of arr) {
|
|
878
|
-
if (
|
|
892
|
+
if (out.length >= sampleLimit) break;
|
|
879
893
|
if (v && typeof v === "object") {
|
|
880
894
|
const row = {};
|
|
881
895
|
for (const [k, val] of Object.entries(v)) row[k] = val == null ? "" : String(val);
|
|
882
|
-
|
|
896
|
+
out.push(row);
|
|
883
897
|
}
|
|
884
898
|
}
|
|
885
|
-
return
|
|
886
|
-
}
|
|
887
|
-
const head = Buffer.alloc(4096);
|
|
888
|
-
const fh = await fs3.open(file, "r");
|
|
889
|
-
let bytesRead = 0;
|
|
890
|
-
try {
|
|
891
|
-
bytesRead = (await fh.read(head, 0, 4096, 0)).bytesRead;
|
|
892
|
-
} finally {
|
|
893
|
-
await fh.close();
|
|
899
|
+
return out;
|
|
894
900
|
}
|
|
895
|
-
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
|
|
917
|
-
|
|
918
|
-
|
|
919
|
-
|
|
920
|
-
|
|
921
|
-
|
|
922
|
-
|
|
923
|
-
|
|
901
|
+
return await new Promise((resolve, reject) => {
|
|
902
|
+
const out = [];
|
|
903
|
+
let done = false;
|
|
904
|
+
const finish = () => {
|
|
905
|
+
if (done) return;
|
|
906
|
+
done = true;
|
|
907
|
+
resolve(out);
|
|
908
|
+
};
|
|
909
|
+
const stream = createReadStream(file);
|
|
910
|
+
const parseStream = Papa.parse(Papa.NODE_STREAM_INPUT, {
|
|
911
|
+
header: true,
|
|
912
|
+
skipEmptyLines: "greedy",
|
|
913
|
+
delimitersToGuess: [",", ";", " ", "|"],
|
|
914
|
+
transformHeader: (h) => h.trim(),
|
|
915
|
+
transform: (v) => typeof v === "string" ? v.trim() : v
|
|
916
|
+
});
|
|
917
|
+
stream.on("error", reject);
|
|
918
|
+
parseStream.on("error", reject);
|
|
919
|
+
parseStream.on("data", (row) => {
|
|
920
|
+
const o = {};
|
|
921
|
+
for (const [k, val] of Object.entries(row)) o[k] = val == null ? "" : String(val);
|
|
922
|
+
out.push(o);
|
|
923
|
+
if (out.length >= sampleLimit) {
|
|
924
|
+
stream.destroy();
|
|
925
|
+
finish();
|
|
926
|
+
}
|
|
927
|
+
});
|
|
928
|
+
parseStream.on("end", finish);
|
|
929
|
+
stream.pipe(parseStream);
|
|
930
|
+
});
|
|
924
931
|
}
|
|
925
932
|
async function buildDryRunReport(file, opts) {
|
|
926
933
|
const sampleLimit = opts.sampleLimit ?? 1e3;
|
|
@@ -1017,7 +1024,7 @@ function trunc(s, n) {
|
|
|
1017
1024
|
|
|
1018
1025
|
// src/index.ts
|
|
1019
1026
|
var program = new Command();
|
|
1020
|
-
program.name("planttaxomatcher").description("PlantTaxoMatcher CLI").version("0.1.
|
|
1027
|
+
program.name("planttaxomatcher").description("PlantTaxoMatcher CLI").version("0.1.1");
|
|
1021
1028
|
program.command("login").description("Save a personal token + server URL").option(
|
|
1022
1029
|
"--token <token>",
|
|
1023
1030
|
"Personal token (ptm_...). Avoid on shared hosts: it is visible in shell history and the process list. Prefer --token-stdin or the PLANTTAXOMATCHER_TOKEN env var."
|
|
@@ -1072,7 +1079,7 @@ program.command("submit <files...>").description(
|
|
|
1072
1079
|
).option(
|
|
1073
1080
|
"--name <label>",
|
|
1074
1081
|
"Job name override (single file only; ignored when multiple files match \u2014 the file name is used)"
|
|
1075
|
-
).requiredOption("--name-column <name>", "Column holding the scientific name").option("--id-column <name>", "Column holding a known ID").option("--family-column <name>", "Column holding family").option("--genus-column <name>", "Column holding genus").option("--rank-column <name>", "Column holding rank").option("--author-column <name>", "Column holding authorship").option("--author-mode <mode>", "ignore|prefer|strict", "prefer").option("--parallel <n>", "Per-job parallelism", "4").option("--review-mode <mode>", "off|recommended|strict", "recommended").option(
|
|
1082
|
+
).requiredOption("--name-column <name>", "Column holding the scientific name").option("--id-column <name>", "Column holding a known ID").option("--family-column <name>", "Column holding family").option("--genus-column <name>", "Column holding genus").option("--rank-column <name>", "Column holding rank").option("--author-column <name>", "Column holding authorship").option("--filter-column <name>", "Only process rows where this column matches --filter-value; others are skipped").option("--filter-value <value>", "Value the --filter-column must equal (trimmed, case-insensitive)").option("--author-mode <mode>", "ignore|prefer|strict", "prefer").option("--parallel <n>", "Per-job parallelism", "4").option("--review-mode <mode>", "off|recommended|strict", "recommended").option(
|
|
1076
1083
|
"--keep-infraspecific",
|
|
1077
1084
|
"Keep infraspecific accepted taxa (varieties, subspecies, forms) instead of collapsing them up to the species",
|
|
1078
1085
|
false
|
|
@@ -1128,6 +1135,8 @@ ${file}`));
|
|
|
1128
1135
|
genusColumn: opts.genusColumn ? String(opts.genusColumn) : null,
|
|
1129
1136
|
rankColumn: opts.rankColumn ? String(opts.rankColumn) : null,
|
|
1130
1137
|
authorColumn: opts.authorColumn ? String(opts.authorColumn) : null,
|
|
1138
|
+
filterColumn: opts.filterColumn ? String(opts.filterColumn) : null,
|
|
1139
|
+
filterValue: opts.filterValue != null ? String(opts.filterValue) : null,
|
|
1131
1140
|
authorMode: String(opts.authorMode ?? "prefer"),
|
|
1132
1141
|
matchAuthors: true,
|
|
1133
1142
|
parallelism: Number(opts.parallel ?? 4),
|
|
@@ -1177,7 +1186,7 @@ program.command("cancel <jobId>").description("Cancel a job. Already-matched row
|
|
|
1177
1186
|
const job = await apiClient.cancelJob(creds, jobId);
|
|
1178
1187
|
console.log(kleur2.green("\u2713"), `cancelled: ${job.id} (status=${job.status})`);
|
|
1179
1188
|
});
|
|
1180
|
-
program.command("download <jobId>").description("Download a job export (CSV, JSON, or NDJSON; optionally bundled with NOTICE.md)").option("--format <fmt>", "csv | json | ndjson", "csv").option("--confirmed-only", "only matched rows that are accepted / not pending", false).option("--bundle", "wrap in a ZIP with NOTICE.md citing the WCVP snapshot + providers", false).option(
|
|
1189
|
+
program.command("download <jobId>").description("Download a job export (CSV, JSON, or NDJSON; optionally bundled with NOTICE.md)").option("--format <fmt>", "csv | xlsx | json | ndjson", "csv").option("--confirmed-only", "only matched rows that are accepted / not pending", false).option("--delimiter <sep>", "CSV separator: comma | semicolon | tab | pipe (csv format only)", "comma").option("--bundle", "wrap in a ZIP with NOTICE.md citing the WCVP snapshot + providers", false).option(
|
|
1181
1190
|
"--columns <list>",
|
|
1182
1191
|
"comma-separated result/upload column keys to KEEP (default: all). See --list-columns"
|
|
1183
1192
|
).option(
|
|
@@ -1198,15 +1207,20 @@ program.command("download <jobId>").description("Download a job export (CSV, JSO
|
|
|
1198
1207
|
return;
|
|
1199
1208
|
}
|
|
1200
1209
|
const format = opts.format ?? "csv";
|
|
1201
|
-
if (format !== "csv" && format !== "json" && format !== "ndjson") {
|
|
1202
|
-
throw new Error(`--format must be csv, json or ndjson (got ${String(format)})`);
|
|
1210
|
+
if (format !== "csv" && format !== "json" && format !== "ndjson" && format !== "xlsx") {
|
|
1211
|
+
throw new Error(`--format must be csv, xlsx, json or ndjson (got ${String(format)})`);
|
|
1203
1212
|
}
|
|
1204
1213
|
const confirmedOnly = !!opts.confirmedOnly;
|
|
1205
1214
|
const bundle = !!opts.bundle;
|
|
1215
|
+
const delimiter = String(opts.delimiter ?? "comma");
|
|
1216
|
+
if (!["comma", "semicolon", "tab", "pipe"].includes(delimiter)) {
|
|
1217
|
+
throw new Error(`--delimiter must be comma, semicolon, tab or pipe (got ${delimiter})`);
|
|
1218
|
+
}
|
|
1206
1219
|
const { filename, body } = await apiClient.downloadJob(creds, jobId, {
|
|
1207
1220
|
format,
|
|
1208
1221
|
confirmedOnly,
|
|
1209
1222
|
bundle,
|
|
1223
|
+
...format === "csv" && delimiter !== "comma" ? { delimiter } : {},
|
|
1210
1224
|
...opts.columns ? { columns: String(opts.columns) } : {},
|
|
1211
1225
|
...opts.wcvpExtra ? { wcvpExtra: String(opts.wcvpExtra) } : {}
|
|
1212
1226
|
});
|
|
@@ -1287,7 +1301,16 @@ async function streamJob(creds, jobId) {
|
|
|
1287
1301
|
const t = String(evt.type ?? "");
|
|
1288
1302
|
if (t === "heartbeat") continue;
|
|
1289
1303
|
if (t === "status") {
|
|
1290
|
-
|
|
1304
|
+
const status = String(evt.status ?? "");
|
|
1305
|
+
console.log(kleur2.gray("\u2022"), status);
|
|
1306
|
+
if (status === "completed") {
|
|
1307
|
+
console.log(kleur2.green("\u2713"), "completed");
|
|
1308
|
+
return;
|
|
1309
|
+
}
|
|
1310
|
+
if (status === "failed" || status === "cancelled") {
|
|
1311
|
+
console.error(kleur2.red(status));
|
|
1312
|
+
return;
|
|
1313
|
+
}
|
|
1291
1314
|
} else if (t === "progress") {
|
|
1292
1315
|
const p = evt.processedQueries ?? evt.processedRows ?? 0;
|
|
1293
1316
|
const total = evt.totalQueries ?? evt.totalRows ?? 0;
|
package/package.json
CHANGED
|
@@ -1,42 +1,44 @@
|
|
|
1
1
|
{
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
2
|
+
"name": "@plantnet/planttaxomatcher",
|
|
3
|
+
"version": "0.1.2",
|
|
4
|
+
"description": "PlantTaxoMatcher CLI — reconcile plant names against WCVP via the PlantTaxoMatcher API.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"engines": {
|
|
7
|
+
"node": ">=24"
|
|
8
|
+
},
|
|
9
|
+
"bin": {
|
|
10
|
+
"planttaxomatcher": "./dist/index.js"
|
|
11
|
+
},
|
|
12
|
+
"files": [
|
|
13
|
+
"dist"
|
|
14
|
+
],
|
|
15
|
+
"publishConfig": {
|
|
16
|
+
"access": "public"
|
|
17
|
+
},
|
|
18
|
+
"scripts": {
|
|
19
|
+
"dev": "tsx src/index.ts",
|
|
20
|
+
"build": "tsup",
|
|
21
|
+
"typecheck": "tsc -p tsconfig.json --noEmit",
|
|
22
|
+
"test": "vitest run --passWithNoTests",
|
|
23
|
+
"lint": "echo \"no lint yet\" && exit 0",
|
|
24
|
+
"prepublishOnly": "pnpm run build"
|
|
25
|
+
},
|
|
26
|
+
"dependencies": {
|
|
27
|
+
"@inquirer/prompts": "^8.5.0",
|
|
28
|
+
"commander": "^14.0.0",
|
|
29
|
+
"kleur": "^4.1.5",
|
|
30
|
+
"ora": "^9.0.0",
|
|
31
|
+
"papaparse": "^5.5.3",
|
|
32
|
+
"undici": "^8.0.0",
|
|
33
|
+
"zod": "^4.0.0"
|
|
34
|
+
},
|
|
35
|
+
"devDependencies": {
|
|
36
|
+
"@planttaxomatcher/shared": "workspace:*",
|
|
37
|
+
"@types/node": "^24.12.4",
|
|
38
|
+
"@types/papaparse": "^5.5.2",
|
|
39
|
+
"tsup": "^8.5.1",
|
|
40
|
+
"tsx": "^4.20.0",
|
|
41
|
+
"typescript": "^6.0.0",
|
|
42
|
+
"vitest": "^4.1.7"
|
|
43
|
+
}
|
|
44
|
+
}
|