flexorch-sdk 0.2.2 → 0.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -24,6 +24,7 @@ npm install flexorch-sdk
24
24
 
25
25
  ```typescript
26
26
  import { FlexOrchClient } from "flexorch-sdk";
27
+ import * as fs from "node:fs/promises";
27
28
 
28
29
  const client = new FlexOrchClient(process.env.FLEXORCH_API_KEY);
29
30
 
@@ -32,7 +33,11 @@ const done = await job.wait();
32
33
 
33
34
  console.log(`Grade: ${done.qualityGrade} Score: ${done.qualityScore}`);
34
35
 
35
- const dataset = await done.dataset();
36
+ // Building a dataset is a separate, explicit step — a completed job doesn't
37
+ // have one until you build it (lets you build one dataset from several jobs,
38
+ // or re-run with forceRebuild: true).
39
+ const built = await done.buildDataset();
40
+ const dataset = await (await built.wait()).dataset();
36
41
  if (dataset) {
37
42
  const bytes = await dataset.export("jsonl");
38
43
  await fs.writeFile("output.jsonl", bytes);
@@ -52,17 +57,27 @@ The API key is read from `FLEXORCH_API_KEY` if not passed explicitly.
52
57
  | Word | `.docx` |
53
58
  | Plain text | `.txt` |
54
59
  | Markdown | `.md` |
60
+ | Spreadsheets | `.xlsx` |
61
+ | Email | `.eml`, `.msg` |
62
+ | Images (OCR) | `.jpg`, `.png`, `.tiff` |
63
+ | Web | `.html`, `.htm` |
55
64
 
56
65
  ---
57
66
 
58
67
  ## Export formats
59
68
 
60
- | Format | Value |
61
- |--------|-------|
62
- | JSON Lines | `"jsonl"` |
63
- | CSV | `"csv"` |
64
- | Parquet | `"parquet"` |
65
- | Excel | `"xlsx"` |
69
+ `"json"` · `"jsonl"` · `"csv"` · `"parquet"` · `"md"` · `"xml"` · `"xlsx"` · `"rag"` · `"hf"`
70
+
71
+ ```typescript
72
+ const bytes = await dataset.export("jsonl");
73
+ await fs.writeFile("output.jsonl", bytes);
74
+
75
+ // rag chunks, only A/B-grade
76
+ const chunks = await dataset.export("rag", { minQuality: "B" });
77
+ ```
78
+
79
+ The `"rag"` format produces LlamaIndex/LangChain-compatible chunks with metadata.
80
+ The `"hf"` format is a zip archive readable with `datasets.load_from_disk()`.
66
81
 
67
82
  ---
68
83
 
@@ -71,13 +86,13 @@ The API key is read from `FLEXORCH_API_KEY` if not passed explicitly.
71
86
  ### `new FlexOrchClient(apiKeyOrOptions?)`
72
87
 
73
88
  ```typescript
74
- const client = new FlexOrchClient("sk-...");
89
+ const client = new FlexOrchClient("dfx_...");
75
90
  // or
76
91
  const client = new FlexOrchClient({
77
- apiKey: "sk-...",
78
- baseUrl: "https://api.flexorch.com", // default
79
- timeout: 30_000, // ms, default 30 s
80
- maxRetries: 3, // default 3
92
+ apiKey: "dfx_...",
93
+ baseUrl: "https://api.flexorch.com/v1", // default
94
+ timeout: 30, // seconds, default 30
95
+ maxRetries: 3, // default 3
81
96
  });
82
97
  ```
83
98
 
@@ -87,7 +102,7 @@ Upload a single file and create a processing job.
87
102
 
88
103
  ```typescript
89
104
  const job = await client.process("invoice.pdf", {
90
- locale: "tr", // BCP-47 locale hint
105
+ locale: "tr", // BCP-47 locale hint; "und" = all PII detectors (default)
91
106
  pipelineConfig: {}, // optional pipeline overrides
92
107
  });
93
108
  ```
@@ -96,7 +111,7 @@ Returns a `Job` instance.
96
111
 
97
112
  ### `client.processMany(files, options?)`
98
113
 
99
- Batch-process multiple files. Returns `Job[]`.
114
+ Batch-process multiple files sequentially. Returns `Job[]`.
100
115
 
101
116
  ```typescript
102
117
  const jobs = await client.processMany(["a.pdf", "b.pdf"], { locale: "en" });
@@ -104,7 +119,7 @@ const jobs = await client.processMany(["a.pdf", "b.pdf"], { locale: "en" });
104
119
 
105
120
  ### `client.processFromS3(connectorId, keys, options?)`
106
121
 
107
- Import files from S3 via a registered connector. Returns `Job[]`.
122
+ Import files from a registered S3 connector. Returns `Job[]`.
108
123
 
109
124
  ```typescript
110
125
  const jobs = await client.processFromS3(conn.id, ["folder/doc.pdf"], { locale: "de" });
@@ -112,12 +127,16 @@ const jobs = await client.processFromS3(conn.id, ["folder/doc.pdf"], { locale: "
112
127
 
113
128
  ### `client.search(query, options?)`
114
129
 
115
- Semantic search across all indexed datasets.
130
+ Semantic search across all indexed datasets (Pro+ plan required).
116
131
 
117
132
  ```typescript
118
- const results = await client.search("net payment terms", { topK: 5 });
133
+ const results = await client.search("net payment terms", {
134
+ topK: 5,
135
+ mode: "auto", // "auto" | "semantic" | "hybrid" | "structured"
136
+ filters: { documentType: "invoice", language: "de", qualityGrade: "A", piiMasked: true },
137
+ });
119
138
  for (const r of results) {
120
- console.log(r.score, r.text);
139
+ console.log(r.score, r.datasetId, r.text);
121
140
  }
122
141
  ```
123
142
 
@@ -128,24 +147,45 @@ for (const r of results) {
128
147
  | Method | Description |
129
148
  |--------|-------------|
130
149
  | `job.wait(options?)` | Poll until done. Resolves with a completed `Job`. |
131
- | `job.dataset()` | Fetch the resulting `Dataset` (null if none). |
150
+ | `job.dataset()` | Fetch the dataset built from this job (`null` if none built yet). |
151
+ | `job.buildDataset(options?)` | Build a dataset from this job's execution. Returns a `dataset_build` `Job` — `wait()` it, then call `.dataset()`. |
152
+
153
+ `wait` options: `{ timeout?: number (seconds), pollInterval?: number (seconds) }`
132
154
 
133
- `wait` options: `{ timeout?: number (seconds), pollInterval?: number (ms) }`
155
+ `buildDataset` options: `{ name?, description?, slug?, forceRebuild?: boolean, replaceExisting?: boolean }`. Throws if the job has no `executionId` (e.g. it failed, or is itself a `dataset_build` job).
134
156
 
135
157
  Throws `JobFailedError` if the job fails, `JobTimeoutError` if timeout is exceeded.
136
158
 
159
+ Key properties: `id`, `status`, `qualityGrade`, `qualityScore`, `documentId`, `executionId`, `hasDataset`, `degraded`, `failureReason`.
160
+
161
+ `degraded` is `true` when the underlying pipeline execution completed but one or more non-critical steps failed (e.g. structured extraction couldn't find a table in the document). The job still succeeds — PII detection and quality scoring results are still meaningful — but the resulting dataset's records/columns may be empty. `wait()` does not throw for a degraded completion.
162
+
137
163
  ---
138
164
 
139
165
  ### `Dataset`
140
166
 
141
167
  | Method | Description |
142
168
  |--------|-------------|
143
- | `dataset.export(format)` | Download dataset as `Buffer`. |
144
- | `dataset.exportToS3(connectorId, format, prefix?)` | Push export to S3. |
145
- | `dataset.index()` | Trigger vector indexing. |
169
+ | `dataset.export(format, options?)` | Download dataset as `Uint8Array`. `options.minQuality` only applies to `format="rag"`. |
170
+ | `dataset.exportToS3(connectorId, format, prefix?)` | Push export directly to S3. |
171
+ | `dataset.rows(options?)` | Preview rows `{ page?, pageSize?, q? }`. |
172
+ | `dataset.profile()` | Quality/privacy profile (grade distribution, PII findings). |
173
+ | `dataset.complianceReport(format?)` | KVKK/GDPR transparency report (Pro+ required). `"json"` (default) or `"pdf"`. |
174
+ | `dataset.index()` | Trigger semantic indexing (Pro+ required). |
146
175
  | `dataset.indexStatus()` | Poll index build status. |
176
+ | `dataset.chunks(options?)` | List RAG chunks (Pro+ required). |
177
+
178
+ Key properties: `id`, `name`, `slug`, `status`, `rowCount`, `createdAt`, `availableFormats`.
147
179
 
148
- Key properties: `id`, `name`, `slug`, `rowCount`, `status`, `qualityGrade`, `qualityScore`, `piiCount`, `createdAt`.
180
+ ---
181
+
182
+ ### `client.documents`
183
+
184
+ ```typescript
185
+ await client.documents.list({ page: 1, pageSize: 20 });
186
+ const doc = await client.documents.get(id); // includes processingHistory, relatedDatasets
187
+ const job = await doc.reprocess(); // re-queue through the pipeline
188
+ ```
149
189
 
150
190
  ---
151
191
 
@@ -167,25 +207,41 @@ await client.connectors.get(id);
167
207
  await client.connectors.delete(id);
168
208
  ```
169
209
 
170
- Supported connector types: `"s3"`. (`"gcs"` and `"azure_blob"` coming in a future release.)
210
+ Connector types: `"s3"` · `"gcs"` · `"azure_blob"` · `"google_drive"` (file sources) and `"pgvector_external"` · `"pinecone"` · `"qdrant"` (vector destinations for dataset indexing).
211
+
212
+ #### Scheduled sync (Pro+)
213
+
214
+ ```typescript
215
+ const schedule = await client.connectors.createSchedule(conn.id, "0 2 * * *", "invoices/");
216
+ await client.connectors.listSchedules(conn.id);
217
+ await client.connectors.triggerSchedule(conn.id, schedule.id); // run now
218
+ await client.connectors.scheduleLogs(conn.id, schedule.id);
219
+ await client.connectors.deleteSchedule(conn.id, schedule.id);
220
+ ```
171
221
 
172
222
  ---
173
223
 
174
224
  ### `client.jobs` / `client.datasets` / `client.usage` / `client.webhooks`
175
225
 
176
226
  ```typescript
177
- await client.jobs.list();
227
+ await client.jobs.list({ page: 1, pageSize: 20 });
178
228
  await client.jobs.get(id);
179
- await client.jobs.cancel(id);
229
+ await client.jobs.submitFeedback(id, "down", { issue: "missing_fields", notes: "PO number not extracted" });
230
+ await client.jobs.getFeedback(id); // null if not submitted
180
231
 
181
232
  await client.datasets.list();
182
233
  await client.datasets.get(id);
183
- await client.datasets.delete(id);
234
+ await client.datasets.buildFromExecution(executionId, { name: "my-dataset" }); // prefer job.buildDataset() if you have a Job
235
+
236
+ const usage = await client.usage.current();
237
+ console.log(`${usage.creditsUsed} / ${usage.creditsLimit} credits used (plan: ${usage.plan})`);
238
+ if (usage.isTrial) console.log(`${usage.trialDaysRemaining} trial days left`);
184
239
 
185
- const snap = await client.usage.current();
186
- // { periodStart, periodEnd, documentsProcessed, documentsLimit, planTier }
240
+ await client.usage.history("30d"); // daily credits + job counts
241
+ await client.usage.qualityTrend("30d"); // daily avg quality score
242
+ await client.usage.rateLimits(); // current window usage, doesn't consume a slot
187
243
 
188
- await client.webhooks.create("https://example.com/hook", ["job.completed"]);
244
+ await client.webhooks.register("https://example.com/hook", ["dataset.ready"]);
189
245
  await client.webhooks.list();
190
246
  await client.webhooks.delete(id);
191
247
  ```
@@ -213,7 +269,7 @@ try {
213
269
  if (err instanceof JobFailedError) {
214
270
  console.error(`Job ${err.jobId} failed: ${err.failureReason}`);
215
271
  } else if (err instanceof QuotaError) {
216
- console.error("Monthly document quota exceeded");
272
+ console.error("Credit limit reached or trial expired");
217
273
  } else if (err instanceof RateLimitError) {
218
274
  console.error(`Rate limited — retry after ${err.retryAfter}s`);
219
275
  } else {
@@ -222,7 +278,7 @@ try {
222
278
  }
223
279
  ```
224
280
 
225
- All SDK errors extend `FlexOrchError`. HTTP 4xx/5xx responses map to typed error classes automatically.
281
+ All SDK errors extend `FlexOrchError`. HTTP 4xx/5xx responses map to typed error classes automatically. `429`/`5xx` are retried with exponential backoff (up to `maxRetries` attempts).
226
282
 
227
283
  ---
228
284
 
@@ -231,8 +287,8 @@ All SDK errors extend `FlexOrchError`. HTTP 4xx/5xx responses map to typed error
231
287
  | Option | Env var | Default |
232
288
  |--------|---------|---------|
233
289
  | `apiKey` | `FLEXORCH_API_KEY` | — |
234
- | `baseUrl` | `FLEXORCH_BASE_URL` | `https://api.flexorch.com` |
235
- | `timeout` | — | `30000` ms |
290
+ | `baseUrl` | | `https://api.flexorch.com/v1` |
291
+ | `timeout` | — | `30` seconds |
236
292
  | `maxRetries` | — | `3` |
237
293
 
238
294
  ---
@@ -241,7 +297,7 @@ All SDK errors extend `FlexOrchError`. HTTP 4xx/5xx responses map to typed error
241
297
 
242
298
  | File | Description |
243
299
  |------|-------------|
244
- | [basic-process.ts](examples/basic-process.ts) | Process a single PDF, export JSONL |
300
+ | [basic-process.ts](examples/basic-process.ts) | Process a single PDF, build a dataset, export JSONL |
245
301
  | [batch-process.ts](examples/batch-process.ts) | Process a directory, collect datasets |
246
302
  | [s3-import.ts](examples/s3-import.ts) | Register S3 connector, import, export back |
247
303
 
@@ -252,7 +308,7 @@ All SDK errors extend `FlexOrchError`. HTTP 4xx/5xx responses map to typed error
252
308
  ```bash
253
309
  npm install
254
310
  npm run build # ESM + CJS + .d.ts
255
- npm test # 36 tests via vitest
311
+ npm test # vitest
256
312
  npm run typecheck # tsc --noEmit
257
313
  ```
258
314
 
@@ -43,11 +43,12 @@ var Dataset = class _Dataset {
43
43
  _transport: transport
44
44
  });
45
45
  }
46
- async export(format) {
46
+ async export(format, opts = {}) {
47
47
  if (!SUPPORTED_FORMATS.has(format)) {
48
48
  throw new Error(`Unsupported format "${format}". Choose from: ${[...SUPPORTED_FORMATS].sort().join(", ")}`);
49
49
  }
50
- return this._transport.getBytes(`/datasets/${this.id}/export`, { format });
50
+ const params = opts.minQuality !== void 0 ? { min_quality: opts.minQuality } : void 0;
51
+ return this._transport.getBytes(`/datasets/${this.id}/export/${format}`, params);
51
52
  }
52
53
  async exportToS3(connectorId, format, prefix = "") {
53
54
  if (!SUPPORTED_FORMATS.has(format)) {
@@ -99,6 +100,25 @@ var Dataset = class _Dataset {
99
100
  totalChunks: Number(data["total_chunks"] ?? 0)
100
101
  };
101
102
  }
103
+ /** Preview dataset rows. */
104
+ async rows(opts = {}) {
105
+ const params = {};
106
+ if (opts.page !== void 0) params["page"] = String(opts.page);
107
+ if (opts.pageSize !== void 0) params["page_size"] = String(opts.pageSize);
108
+ if (opts.q !== void 0) params["q"] = opts.q;
109
+ return await this._transport.get(`/datasets/${this.id}/rows`, params) ?? {};
110
+ }
111
+ /** Quality/privacy profile — only available once status is "ready". */
112
+ async profile() {
113
+ return await this._transport.get(`/datasets/${this.id}/profile`) ?? {};
114
+ }
115
+ /** KVKK/GDPR processing transparency report (Pro+ required). */
116
+ async complianceReport(format = "json") {
117
+ if (format === "pdf") {
118
+ return this._transport.getBytes(`/datasets/${this.id}/compliance-report`, { format: "pdf" });
119
+ }
120
+ return await this._transport.get(`/datasets/${this.id}/compliance-report`, { format }) ?? {};
121
+ }
102
122
  toString() {
103
123
  return `Dataset(id=${this.id}, name=${this.name}, rows=${this.rowCount}, status=${this.status})`;
104
124
  }
@@ -1,6 +1,6 @@
1
1
  import {
2
2
  Dataset
3
- } from "./chunk-35RGZSFP.js";
3
+ } from "./chunk-4KCMMRD7.js";
4
4
  export {
5
5
  Dataset
6
6
  };