@plantnet/planttaxomatcher 0.1.0 → 0.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (2) hide show
  1. package/dist/index.js +87 -64
  2. package/package.json +43 -41
package/dist/index.js CHANGED
@@ -101,7 +101,16 @@ var jobConfigSchema = z3.object({
101
101
  * resolved version on `jobs.referential_version` so re-runs are
102
102
  * reproducible even if a newer snapshot lands later.
103
103
  */
104
- referentialVersion: z3.string().min(1).optional()
104
+ referentialVersion: z3.string().min(1).optional(),
105
+ /**
106
+ * Row filter: keep only rows whose `filterColumn` value equals
107
+ * `filterValue` (trimmed, case-insensitive); all other rows are skipped at
108
+ * ingest (no match query created). Both must be set for the filter to
109
+ * apply. Lets a user match a subset of a mixed file (e.g. only
110
+ * `kingdom = Plantae`).
111
+ */
112
+ filterColumn: z3.string().nullable().optional(),
113
+ filterValue: z3.string().nullable().optional()
105
114
  });
106
115
  var wcvpSnapshotSummarySchema = z3.object({
107
116
  version: z3.string(),
@@ -785,6 +794,7 @@ var apiClient = {
785
794
  const params = new URLSearchParams({ format: opts.format });
786
795
  if (opts.confirmedOnly) params.set("confirmedOnly", "true");
787
796
  if (opts.bundle) params.set("bundle", "true");
797
+ if (opts.delimiter) params.set("delimiter", opts.delimiter);
788
798
  if (opts.columns) params.set("columns", opts.columns);
789
799
  if (opts.wcvpExtra) params.set("wcvpExtra", opts.wcvpExtra);
790
800
  const url = new URL(`/v1/jobs/${id}/download?${params}`, creds.server).toString();
@@ -795,7 +805,7 @@ var apiClient = {
795
805
  }
796
806
  const disp = res.headers.get("content-disposition") ?? "";
797
807
  const m = /filename="?([^"]+)"?/.exec(disp);
798
- const ext = opts.bundle ? "zip" : opts.format === "csv" ? "csv" : opts.format === "json" ? "json" : "ndjson";
808
+ const ext = opts.bundle ? "zip" : opts.format === "csv" ? "csv" : opts.format === "json" ? "json" : opts.format === "xlsx" ? "xlsx" : "ndjson";
799
809
  const filename = m?.[1] ?? `planttaxomatcher_${id}.${ext}`;
800
810
  const buf = Buffer.from(await res.arrayBuffer());
801
811
  return { filename, body: buf };
@@ -810,20 +820,25 @@ var apiClient = {
810
820
  const reader = res.body.getReader();
811
821
  const decoder = new TextDecoder("utf-8");
812
822
  let buf = "";
813
- for (; ; ) {
814
- const { done, value } = await reader.read();
815
- if (done) break;
816
- buf += decoder.decode(value, { stream: true });
817
- let nl;
818
- while ((nl = buf.indexOf("\n")) >= 0) {
819
- const line = buf.slice(0, nl).trim();
820
- buf = buf.slice(nl + 1);
821
- if (!line) continue;
822
- try {
823
- yield JSON.parse(line);
824
- } catch {
823
+ try {
824
+ for (; ; ) {
825
+ const { done, value } = await reader.read();
826
+ if (done) break;
827
+ buf += decoder.decode(value, { stream: true });
828
+ let nl;
829
+ while ((nl = buf.indexOf("\n")) >= 0) {
830
+ const line = buf.slice(0, nl).trim();
831
+ buf = buf.slice(nl + 1);
832
+ if (!line) continue;
833
+ try {
834
+ yield JSON.parse(line);
835
+ } catch {
836
+ }
825
837
  }
826
838
  }
839
+ } finally {
840
+ await reader.cancel().catch(() => {
841
+ });
827
842
  }
828
843
  }
829
844
  };
@@ -863,9 +878,8 @@ async function requireCredentials() {
863
878
  }
864
879
 
865
880
  // src/dry-run.ts
866
- import { promises as fs3 } from "fs";
867
- import { Readable } from "stream";
868
- import { parse as parseCsvStream } from "csv-parse";
881
+ import { promises as fs3, createReadStream } from "fs";
882
+ import Papa from "papaparse";
869
883
  import kleur from "kleur";
870
884
  async function readSample(file, sampleLimit) {
871
885
  const lower = file.toLowerCase();
@@ -873,54 +887,47 @@ async function readSample(file, sampleLimit) {
873
887
  const text = await fs3.readFile(file, "utf8");
874
888
  const arr = JSON.parse(text);
875
889
  if (!Array.isArray(arr)) throw new Error("JSON file must be an array of row objects");
876
- const out2 = [];
890
+ const out = [];
877
891
  for (const v of arr) {
878
- if (out2.length >= sampleLimit) break;
892
+ if (out.length >= sampleLimit) break;
879
893
  if (v && typeof v === "object") {
880
894
  const row = {};
881
895
  for (const [k, val] of Object.entries(v)) row[k] = val == null ? "" : String(val);
882
- out2.push(row);
896
+ out.push(row);
883
897
  }
884
898
  }
885
- return out2;
886
- }
887
- const head = Buffer.alloc(4096);
888
- const fh = await fs3.open(file, "r");
889
- let bytesRead = 0;
890
- try {
891
- bytesRead = (await fh.read(head, 0, 4096, 0)).bytesRead;
892
- } finally {
893
- await fh.close();
899
+ return out;
894
900
  }
895
- const delimiter = detectDelimiter(head.subarray(0, bytesRead));
896
- const buf = await fs3.readFile(file);
897
- const parser = Readable.from(buf).pipe(
898
- parseCsvStream({
899
- columns: true,
900
- skip_empty_lines: true,
901
- trim: true,
902
- delimiter,
903
- relax_quotes: true,
904
- relax_column_count: true
905
- })
906
- );
907
- const out = [];
908
- for await (const row of parser) {
909
- out.push(row);
910
- if (out.length >= sampleLimit) break;
911
- }
912
- return out;
913
- }
914
- function detectDelimiter(buf) {
915
- const head = buf.toString("utf8").split("\n", 5).join("\n");
916
- const counts = {
917
- ",": (head.match(/,/g) ?? []).length,
918
- ";": (head.match(/;/g) ?? []).length,
919
- " ": (head.match(/\t/g) ?? []).length,
920
- "|": (head.match(/\|/g) ?? []).length
921
- };
922
- const winner = Object.entries(counts).sort((a, b) => b[1] - a[1])[0];
923
- return winner && winner[1] > 0 ? winner[0] : ",";
901
+ return await new Promise((resolve, reject) => {
902
+ const out = [];
903
+ let done = false;
904
+ const finish = () => {
905
+ if (done) return;
906
+ done = true;
907
+ resolve(out);
908
+ };
909
+ const stream = createReadStream(file);
910
+ const parseStream = Papa.parse(Papa.NODE_STREAM_INPUT, {
911
+ header: true,
912
+ skipEmptyLines: "greedy",
913
+ delimitersToGuess: [",", ";", " ", "|"],
914
+ transformHeader: (h) => h.trim(),
915
+ transform: (v) => typeof v === "string" ? v.trim() : v
916
+ });
917
+ stream.on("error", reject);
918
+ parseStream.on("error", reject);
919
+ parseStream.on("data", (row) => {
920
+ const o = {};
921
+ for (const [k, val] of Object.entries(row)) o[k] = val == null ? "" : String(val);
922
+ out.push(o);
923
+ if (out.length >= sampleLimit) {
924
+ stream.destroy();
925
+ finish();
926
+ }
927
+ });
928
+ parseStream.on("end", finish);
929
+ stream.pipe(parseStream);
930
+ });
924
931
  }
925
932
  async function buildDryRunReport(file, opts) {
926
933
  const sampleLimit = opts.sampleLimit ?? 1e3;
@@ -1017,7 +1024,7 @@ function trunc(s, n) {
1017
1024
 
1018
1025
  // src/index.ts
1019
1026
  var program = new Command();
1020
- program.name("planttaxomatcher").description("PlantTaxoMatcher CLI").version("0.1.0");
1027
+ program.name("planttaxomatcher").description("PlantTaxoMatcher CLI").version("0.1.1");
1021
1028
  program.command("login").description("Save a personal token + server URL").option(
1022
1029
  "--token <token>",
1023
1030
  "Personal token (ptm_...). Avoid on shared hosts: it is visible in shell history and the process list. Prefer --token-stdin or the PLANTTAXOMATCHER_TOKEN env var."
@@ -1072,7 +1079,7 @@ program.command("submit <files...>").description(
1072
1079
  ).option(
1073
1080
  "--name <label>",
1074
1081
  "Job name override (single file only; ignored when multiple files match \u2014 the file name is used)"
1075
- ).requiredOption("--name-column <name>", "Column holding the scientific name").option("--id-column <name>", "Column holding a known ID").option("--family-column <name>", "Column holding family").option("--genus-column <name>", "Column holding genus").option("--rank-column <name>", "Column holding rank").option("--author-column <name>", "Column holding authorship").option("--author-mode <mode>", "ignore|prefer|strict", "prefer").option("--parallel <n>", "Per-job parallelism", "4").option("--review-mode <mode>", "off|recommended|strict", "recommended").option(
1082
+ ).requiredOption("--name-column <name>", "Column holding the scientific name").option("--id-column <name>", "Column holding a known ID").option("--family-column <name>", "Column holding family").option("--genus-column <name>", "Column holding genus").option("--rank-column <name>", "Column holding rank").option("--author-column <name>", "Column holding authorship").option("--filter-column <name>", "Only process rows where this column matches --filter-value; others are skipped").option("--filter-value <value>", "Value the --filter-column must equal (trimmed, case-insensitive)").option("--author-mode <mode>", "ignore|prefer|strict", "prefer").option("--parallel <n>", "Per-job parallelism", "4").option("--review-mode <mode>", "off|recommended|strict", "recommended").option(
1076
1083
  "--keep-infraspecific",
1077
1084
  "Keep infraspecific accepted taxa (varieties, subspecies, forms) instead of collapsing them up to the species",
1078
1085
  false
@@ -1128,6 +1135,8 @@ ${file}`));
1128
1135
  genusColumn: opts.genusColumn ? String(opts.genusColumn) : null,
1129
1136
  rankColumn: opts.rankColumn ? String(opts.rankColumn) : null,
1130
1137
  authorColumn: opts.authorColumn ? String(opts.authorColumn) : null,
1138
+ filterColumn: opts.filterColumn ? String(opts.filterColumn) : null,
1139
+ filterValue: opts.filterValue != null ? String(opts.filterValue) : null,
1131
1140
  authorMode: String(opts.authorMode ?? "prefer"),
1132
1141
  matchAuthors: true,
1133
1142
  parallelism: Number(opts.parallel ?? 4),
@@ -1177,7 +1186,7 @@ program.command("cancel <jobId>").description("Cancel a job. Already-matched row
1177
1186
  const job = await apiClient.cancelJob(creds, jobId);
1178
1187
  console.log(kleur2.green("\u2713"), `cancelled: ${job.id} (status=${job.status})`);
1179
1188
  });
1180
- program.command("download <jobId>").description("Download a job export (CSV, JSON, or NDJSON; optionally bundled with NOTICE.md)").option("--format <fmt>", "csv | json | ndjson", "csv").option("--confirmed-only", "only matched rows that are accepted / not pending", false).option("--bundle", "wrap in a ZIP with NOTICE.md citing the WCVP snapshot + providers", false).option(
1189
+ program.command("download <jobId>").description("Download a job export (CSV, JSON, or NDJSON; optionally bundled with NOTICE.md)").option("--format <fmt>", "csv | xlsx | json | ndjson", "csv").option("--confirmed-only", "only matched rows that are accepted / not pending", false).option("--delimiter <sep>", "CSV separator: comma | semicolon | tab | pipe (csv format only)", "comma").option("--bundle", "wrap in a ZIP with NOTICE.md citing the WCVP snapshot + providers", false).option(
1181
1190
  "--columns <list>",
1182
1191
  "comma-separated result/upload column keys to KEEP (default: all). See --list-columns"
1183
1192
  ).option(
@@ -1198,15 +1207,20 @@ program.command("download <jobId>").description("Download a job export (CSV, JSO
1198
1207
  return;
1199
1208
  }
1200
1209
  const format = opts.format ?? "csv";
1201
- if (format !== "csv" && format !== "json" && format !== "ndjson") {
1202
- throw new Error(`--format must be csv, json or ndjson (got ${String(format)})`);
1210
+ if (format !== "csv" && format !== "json" && format !== "ndjson" && format !== "xlsx") {
1211
+ throw new Error(`--format must be csv, xlsx, json or ndjson (got ${String(format)})`);
1203
1212
  }
1204
1213
  const confirmedOnly = !!opts.confirmedOnly;
1205
1214
  const bundle = !!opts.bundle;
1215
+ const delimiter = String(opts.delimiter ?? "comma");
1216
+ if (!["comma", "semicolon", "tab", "pipe"].includes(delimiter)) {
1217
+ throw new Error(`--delimiter must be comma, semicolon, tab or pipe (got ${delimiter})`);
1218
+ }
1206
1219
  const { filename, body } = await apiClient.downloadJob(creds, jobId, {
1207
1220
  format,
1208
1221
  confirmedOnly,
1209
1222
  bundle,
1223
+ ...format === "csv" && delimiter !== "comma" ? { delimiter } : {},
1210
1224
  ...opts.columns ? { columns: String(opts.columns) } : {},
1211
1225
  ...opts.wcvpExtra ? { wcvpExtra: String(opts.wcvpExtra) } : {}
1212
1226
  });
@@ -1287,7 +1301,16 @@ async function streamJob(creds, jobId) {
1287
1301
  const t = String(evt.type ?? "");
1288
1302
  if (t === "heartbeat") continue;
1289
1303
  if (t === "status") {
1290
- console.log(kleur2.gray("\u2022"), evt.status);
1304
+ const status = String(evt.status ?? "");
1305
+ console.log(kleur2.gray("\u2022"), status);
1306
+ if (status === "completed") {
1307
+ console.log(kleur2.green("\u2713"), "completed");
1308
+ return;
1309
+ }
1310
+ if (status === "failed" || status === "cancelled") {
1311
+ console.error(kleur2.red(status));
1312
+ return;
1313
+ }
1291
1314
  } else if (t === "progress") {
1292
1315
  const p = evt.processedQueries ?? evt.processedRows ?? 0;
1293
1316
  const total = evt.totalQueries ?? evt.totalRows ?? 0;
package/package.json CHANGED
@@ -1,42 +1,44 @@
1
1
  {
2
- "name": "@plantnet/planttaxomatcher",
3
- "version": "0.1.0",
4
- "description": "PlantTaxoMatcher CLI — reconcile plant names against WCVP via the PlantTaxoMatcher API.",
5
- "type": "module",
6
- "engines": {
7
- "node": ">=24"
8
- },
9
- "bin": {
10
- "planttaxomatcher": "./dist/index.js"
11
- },
12
- "files": [
13
- "dist"
14
- ],
15
- "publishConfig": {
16
- "access": "public"
17
- },
18
- "dependencies": {
19
- "@inquirer/prompts": "^8.5.0",
20
- "commander": "^14.0.0",
21
- "csv-parse": "^6.0.0",
22
- "kleur": "^4.1.5",
23
- "ora": "^9.0.0",
24
- "undici": "^8.0.0",
25
- "zod": "^4.0.0"
26
- },
27
- "devDependencies": {
28
- "@types/node": "^24.12.4",
29
- "tsup": "^8.5.1",
30
- "tsx": "^4.20.0",
31
- "typescript": "^6.0.0",
32
- "vitest": "^4.1.7",
33
- "@planttaxomatcher/shared": "0.0.0"
34
- },
35
- "scripts": {
36
- "dev": "tsx src/index.ts",
37
- "build": "tsup",
38
- "typecheck": "tsc -p tsconfig.json --noEmit",
39
- "test": "vitest run --passWithNoTests",
40
- "lint": "echo \"no lint yet\" && exit 0"
41
- }
42
- }
2
+ "name": "@plantnet/planttaxomatcher",
3
+ "version": "0.1.2",
4
+ "description": "PlantTaxoMatcher CLI — reconcile plant names against WCVP via the PlantTaxoMatcher API.",
5
+ "type": "module",
6
+ "engines": {
7
+ "node": ">=24"
8
+ },
9
+ "bin": {
10
+ "planttaxomatcher": "./dist/index.js"
11
+ },
12
+ "files": [
13
+ "dist"
14
+ ],
15
+ "publishConfig": {
16
+ "access": "public"
17
+ },
18
+ "scripts": {
19
+ "dev": "tsx src/index.ts",
20
+ "build": "tsup",
21
+ "typecheck": "tsc -p tsconfig.json --noEmit",
22
+ "test": "vitest run --passWithNoTests",
23
+ "lint": "echo \"no lint yet\" && exit 0",
24
+ "prepublishOnly": "pnpm run build"
25
+ },
26
+ "dependencies": {
27
+ "@inquirer/prompts": "^8.5.0",
28
+ "commander": "^14.0.0",
29
+ "kleur": "^4.1.5",
30
+ "ora": "^9.0.0",
31
+ "papaparse": "^5.5.3",
32
+ "undici": "^8.0.0",
33
+ "zod": "^4.0.0"
34
+ },
35
+ "devDependencies": {
36
+ "@planttaxomatcher/shared": "workspace:*",
37
+ "@types/node": "^24.12.4",
38
+ "@types/papaparse": "^5.5.2",
39
+ "tsup": "^8.5.1",
40
+ "tsx": "^4.20.0",
41
+ "typescript": "^6.0.0",
42
+ "vitest": "^4.1.7"
43
+ }
44
+ }