@plantnet/planttaxomatcher 0.1.0 → 0.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.js +74 -62
- package/package.json +43 -41
package/dist/index.js
CHANGED
|
@@ -785,6 +785,7 @@ var apiClient = {
|
|
|
785
785
|
const params = new URLSearchParams({ format: opts.format });
|
|
786
786
|
if (opts.confirmedOnly) params.set("confirmedOnly", "true");
|
|
787
787
|
if (opts.bundle) params.set("bundle", "true");
|
|
788
|
+
if (opts.delimiter) params.set("delimiter", opts.delimiter);
|
|
788
789
|
if (opts.columns) params.set("columns", opts.columns);
|
|
789
790
|
if (opts.wcvpExtra) params.set("wcvpExtra", opts.wcvpExtra);
|
|
790
791
|
const url = new URL(`/v1/jobs/${id}/download?${params}`, creds.server).toString();
|
|
@@ -795,7 +796,7 @@ var apiClient = {
|
|
|
795
796
|
}
|
|
796
797
|
const disp = res.headers.get("content-disposition") ?? "";
|
|
797
798
|
const m = /filename="?([^"]+)"?/.exec(disp);
|
|
798
|
-
const ext = opts.bundle ? "zip" : opts.format === "csv" ? "csv" : opts.format === "json" ? "json" : "ndjson";
|
|
799
|
+
const ext = opts.bundle ? "zip" : opts.format === "csv" ? "csv" : opts.format === "json" ? "json" : opts.format === "xlsx" ? "xlsx" : "ndjson";
|
|
799
800
|
const filename = m?.[1] ?? `planttaxomatcher_${id}.${ext}`;
|
|
800
801
|
const buf = Buffer.from(await res.arrayBuffer());
|
|
801
802
|
return { filename, body: buf };
|
|
@@ -810,20 +811,25 @@ var apiClient = {
|
|
|
810
811
|
const reader = res.body.getReader();
|
|
811
812
|
const decoder = new TextDecoder("utf-8");
|
|
812
813
|
let buf = "";
|
|
813
|
-
|
|
814
|
-
|
|
815
|
-
|
|
816
|
-
|
|
817
|
-
|
|
818
|
-
|
|
819
|
-
|
|
820
|
-
|
|
821
|
-
|
|
822
|
-
|
|
823
|
-
|
|
824
|
-
|
|
814
|
+
try {
|
|
815
|
+
for (; ; ) {
|
|
816
|
+
const { done, value } = await reader.read();
|
|
817
|
+
if (done) break;
|
|
818
|
+
buf += decoder.decode(value, { stream: true });
|
|
819
|
+
let nl;
|
|
820
|
+
while ((nl = buf.indexOf("\n")) >= 0) {
|
|
821
|
+
const line = buf.slice(0, nl).trim();
|
|
822
|
+
buf = buf.slice(nl + 1);
|
|
823
|
+
if (!line) continue;
|
|
824
|
+
try {
|
|
825
|
+
yield JSON.parse(line);
|
|
826
|
+
} catch {
|
|
827
|
+
}
|
|
825
828
|
}
|
|
826
829
|
}
|
|
830
|
+
} finally {
|
|
831
|
+
await reader.cancel().catch(() => {
|
|
832
|
+
});
|
|
827
833
|
}
|
|
828
834
|
}
|
|
829
835
|
};
|
|
@@ -863,9 +869,8 @@ async function requireCredentials() {
|
|
|
863
869
|
}
|
|
864
870
|
|
|
865
871
|
// src/dry-run.ts
|
|
866
|
-
import { promises as fs3 } from "fs";
|
|
867
|
-
import
|
|
868
|
-
import { parse as parseCsvStream } from "csv-parse";
|
|
872
|
+
import { promises as fs3, createReadStream } from "fs";
|
|
873
|
+
import Papa from "papaparse";
|
|
869
874
|
import kleur from "kleur";
|
|
870
875
|
async function readSample(file, sampleLimit) {
|
|
871
876
|
const lower = file.toLowerCase();
|
|
@@ -873,54 +878,47 @@ async function readSample(file, sampleLimit) {
|
|
|
873
878
|
const text = await fs3.readFile(file, "utf8");
|
|
874
879
|
const arr = JSON.parse(text);
|
|
875
880
|
if (!Array.isArray(arr)) throw new Error("JSON file must be an array of row objects");
|
|
876
|
-
const
|
|
881
|
+
const out = [];
|
|
877
882
|
for (const v of arr) {
|
|
878
|
-
if (
|
|
883
|
+
if (out.length >= sampleLimit) break;
|
|
879
884
|
if (v && typeof v === "object") {
|
|
880
885
|
const row = {};
|
|
881
886
|
for (const [k, val] of Object.entries(v)) row[k] = val == null ? "" : String(val);
|
|
882
|
-
|
|
887
|
+
out.push(row);
|
|
883
888
|
}
|
|
884
889
|
}
|
|
885
|
-
return
|
|
886
|
-
}
|
|
887
|
-
const head = Buffer.alloc(4096);
|
|
888
|
-
const fh = await fs3.open(file, "r");
|
|
889
|
-
let bytesRead = 0;
|
|
890
|
-
try {
|
|
891
|
-
bytesRead = (await fh.read(head, 0, 4096, 0)).bytesRead;
|
|
892
|
-
} finally {
|
|
893
|
-
await fh.close();
|
|
890
|
+
return out;
|
|
894
891
|
}
|
|
895
|
-
|
|
896
|
-
|
|
897
|
-
|
|
898
|
-
|
|
899
|
-
|
|
900
|
-
|
|
901
|
-
|
|
902
|
-
|
|
903
|
-
|
|
904
|
-
|
|
905
|
-
|
|
906
|
-
|
|
907
|
-
|
|
908
|
-
|
|
909
|
-
|
|
910
|
-
|
|
911
|
-
|
|
912
|
-
|
|
913
|
-
|
|
914
|
-
|
|
915
|
-
|
|
916
|
-
|
|
917
|
-
|
|
918
|
-
|
|
919
|
-
|
|
920
|
-
|
|
921
|
-
|
|
922
|
-
|
|
923
|
-
|
|
892
|
+
return await new Promise((resolve, reject) => {
|
|
893
|
+
const out = [];
|
|
894
|
+
let done = false;
|
|
895
|
+
const finish = () => {
|
|
896
|
+
if (done) return;
|
|
897
|
+
done = true;
|
|
898
|
+
resolve(out);
|
|
899
|
+
};
|
|
900
|
+
const stream = createReadStream(file);
|
|
901
|
+
const parseStream = Papa.parse(Papa.NODE_STREAM_INPUT, {
|
|
902
|
+
header: true,
|
|
903
|
+
skipEmptyLines: "greedy",
|
|
904
|
+
delimitersToGuess: [",", ";", " ", "|"],
|
|
905
|
+
transformHeader: (h) => h.trim(),
|
|
906
|
+
transform: (v) => typeof v === "string" ? v.trim() : v
|
|
907
|
+
});
|
|
908
|
+
stream.on("error", reject);
|
|
909
|
+
parseStream.on("error", reject);
|
|
910
|
+
parseStream.on("data", (row) => {
|
|
911
|
+
const o = {};
|
|
912
|
+
for (const [k, val] of Object.entries(row)) o[k] = val == null ? "" : String(val);
|
|
913
|
+
out.push(o);
|
|
914
|
+
if (out.length >= sampleLimit) {
|
|
915
|
+
stream.destroy();
|
|
916
|
+
finish();
|
|
917
|
+
}
|
|
918
|
+
});
|
|
919
|
+
parseStream.on("end", finish);
|
|
920
|
+
stream.pipe(parseStream);
|
|
921
|
+
});
|
|
924
922
|
}
|
|
925
923
|
async function buildDryRunReport(file, opts) {
|
|
926
924
|
const sampleLimit = opts.sampleLimit ?? 1e3;
|
|
@@ -1017,7 +1015,7 @@ function trunc(s, n) {
|
|
|
1017
1015
|
|
|
1018
1016
|
// src/index.ts
|
|
1019
1017
|
var program = new Command();
|
|
1020
|
-
program.name("planttaxomatcher").description("PlantTaxoMatcher CLI").version("0.1.
|
|
1018
|
+
program.name("planttaxomatcher").description("PlantTaxoMatcher CLI").version("0.1.1");
|
|
1021
1019
|
program.command("login").description("Save a personal token + server URL").option(
|
|
1022
1020
|
"--token <token>",
|
|
1023
1021
|
"Personal token (ptm_...). Avoid on shared hosts: it is visible in shell history and the process list. Prefer --token-stdin or the PLANTTAXOMATCHER_TOKEN env var."
|
|
@@ -1177,7 +1175,7 @@ program.command("cancel <jobId>").description("Cancel a job. Already-matched row
|
|
|
1177
1175
|
const job = await apiClient.cancelJob(creds, jobId);
|
|
1178
1176
|
console.log(kleur2.green("\u2713"), `cancelled: ${job.id} (status=${job.status})`);
|
|
1179
1177
|
});
|
|
1180
|
-
program.command("download <jobId>").description("Download a job export (CSV, JSON, or NDJSON; optionally bundled with NOTICE.md)").option("--format <fmt>", "csv | json | ndjson", "csv").option("--confirmed-only", "only matched rows that are accepted / not pending", false).option("--bundle", "wrap in a ZIP with NOTICE.md citing the WCVP snapshot + providers", false).option(
|
|
1178
|
+
program.command("download <jobId>").description("Download a job export (CSV, JSON, or NDJSON; optionally bundled with NOTICE.md)").option("--format <fmt>", "csv | xlsx | json | ndjson", "csv").option("--confirmed-only", "only matched rows that are accepted / not pending", false).option("--delimiter <sep>", "CSV separator: comma | semicolon | tab | pipe (csv format only)", "comma").option("--bundle", "wrap in a ZIP with NOTICE.md citing the WCVP snapshot + providers", false).option(
|
|
1181
1179
|
"--columns <list>",
|
|
1182
1180
|
"comma-separated result/upload column keys to KEEP (default: all). See --list-columns"
|
|
1183
1181
|
).option(
|
|
@@ -1198,15 +1196,20 @@ program.command("download <jobId>").description("Download a job export (CSV, JSO
|
|
|
1198
1196
|
return;
|
|
1199
1197
|
}
|
|
1200
1198
|
const format = opts.format ?? "csv";
|
|
1201
|
-
if (format !== "csv" && format !== "json" && format !== "ndjson") {
|
|
1202
|
-
throw new Error(`--format must be csv, json or ndjson (got ${String(format)})`);
|
|
1199
|
+
if (format !== "csv" && format !== "json" && format !== "ndjson" && format !== "xlsx") {
|
|
1200
|
+
throw new Error(`--format must be csv, xlsx, json or ndjson (got ${String(format)})`);
|
|
1203
1201
|
}
|
|
1204
1202
|
const confirmedOnly = !!opts.confirmedOnly;
|
|
1205
1203
|
const bundle = !!opts.bundle;
|
|
1204
|
+
const delimiter = String(opts.delimiter ?? "comma");
|
|
1205
|
+
if (!["comma", "semicolon", "tab", "pipe"].includes(delimiter)) {
|
|
1206
|
+
throw new Error(`--delimiter must be comma, semicolon, tab or pipe (got ${delimiter})`);
|
|
1207
|
+
}
|
|
1206
1208
|
const { filename, body } = await apiClient.downloadJob(creds, jobId, {
|
|
1207
1209
|
format,
|
|
1208
1210
|
confirmedOnly,
|
|
1209
1211
|
bundle,
|
|
1212
|
+
...format === "csv" && delimiter !== "comma" ? { delimiter } : {},
|
|
1210
1213
|
...opts.columns ? { columns: String(opts.columns) } : {},
|
|
1211
1214
|
...opts.wcvpExtra ? { wcvpExtra: String(opts.wcvpExtra) } : {}
|
|
1212
1215
|
});
|
|
@@ -1287,7 +1290,16 @@ async function streamJob(creds, jobId) {
|
|
|
1287
1290
|
const t = String(evt.type ?? "");
|
|
1288
1291
|
if (t === "heartbeat") continue;
|
|
1289
1292
|
if (t === "status") {
|
|
1290
|
-
|
|
1293
|
+
const status = String(evt.status ?? "");
|
|
1294
|
+
console.log(kleur2.gray("\u2022"), status);
|
|
1295
|
+
if (status === "completed") {
|
|
1296
|
+
console.log(kleur2.green("\u2713"), "completed");
|
|
1297
|
+
return;
|
|
1298
|
+
}
|
|
1299
|
+
if (status === "failed" || status === "cancelled") {
|
|
1300
|
+
console.error(kleur2.red(status));
|
|
1301
|
+
return;
|
|
1302
|
+
}
|
|
1291
1303
|
} else if (t === "progress") {
|
|
1292
1304
|
const p = evt.processedQueries ?? evt.processedRows ?? 0;
|
|
1293
1305
|
const total = evt.totalQueries ?? evt.totalRows ?? 0;
|
package/package.json
CHANGED
|
@@ -1,42 +1,44 @@
|
|
|
1
1
|
{
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
5
|
-
|
|
6
|
-
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
25
|
-
|
|
26
|
-
|
|
27
|
-
|
|
28
|
-
|
|
29
|
-
|
|
30
|
-
|
|
31
|
-
|
|
32
|
-
|
|
33
|
-
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
|
|
40
|
-
|
|
41
|
-
|
|
42
|
-
|
|
2
|
+
"name": "@plantnet/planttaxomatcher",
|
|
3
|
+
"version": "0.1.1",
|
|
4
|
+
"description": "PlantTaxoMatcher CLI — reconcile plant names against WCVP via the PlantTaxoMatcher API.",
|
|
5
|
+
"type": "module",
|
|
6
|
+
"engines": {
|
|
7
|
+
"node": ">=24"
|
|
8
|
+
},
|
|
9
|
+
"bin": {
|
|
10
|
+
"planttaxomatcher": "./dist/index.js"
|
|
11
|
+
},
|
|
12
|
+
"files": [
|
|
13
|
+
"dist"
|
|
14
|
+
],
|
|
15
|
+
"publishConfig": {
|
|
16
|
+
"access": "public"
|
|
17
|
+
},
|
|
18
|
+
"scripts": {
|
|
19
|
+
"dev": "tsx src/index.ts",
|
|
20
|
+
"build": "tsup",
|
|
21
|
+
"typecheck": "tsc -p tsconfig.json --noEmit",
|
|
22
|
+
"test": "vitest run --passWithNoTests",
|
|
23
|
+
"lint": "echo \"no lint yet\" && exit 0",
|
|
24
|
+
"prepublishOnly": "pnpm run build"
|
|
25
|
+
},
|
|
26
|
+
"dependencies": {
|
|
27
|
+
"@inquirer/prompts": "^8.5.0",
|
|
28
|
+
"commander": "^14.0.0",
|
|
29
|
+
"kleur": "^4.1.5",
|
|
30
|
+
"ora": "^9.0.0",
|
|
31
|
+
"papaparse": "^5.5.3",
|
|
32
|
+
"undici": "^8.0.0",
|
|
33
|
+
"zod": "^4.0.0"
|
|
34
|
+
},
|
|
35
|
+
"devDependencies": {
|
|
36
|
+
"@planttaxomatcher/shared": "workspace:*",
|
|
37
|
+
"@types/node": "^24.12.4",
|
|
38
|
+
"@types/papaparse": "^5.5.2",
|
|
39
|
+
"tsup": "^8.5.1",
|
|
40
|
+
"tsx": "^4.20.0",
|
|
41
|
+
"typescript": "^6.0.0",
|
|
42
|
+
"vitest": "^4.1.7"
|
|
43
|
+
}
|
|
44
|
+
}
|