runbios-sdk 0.2.1-rc.143 → 0.2.1-rc.151

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/client.d.ts CHANGED
@@ -1,4 +1,4 @@
1
- import type { BiOSConfig, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement } from './types.js';
1
+ import type { BiOSConfig, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, LoopImportResult } from './types.js';
2
2
  /**
3
3
  * Typed error thrown by every SDK method when the API returns a non-2xx status.
4
4
  *
@@ -135,6 +135,53 @@ export declare const COMING_SOON_CODES: readonly ["TRAINING_COMING_SOON", "DATAS
135
135
  export declare class ComingSoonError extends ApiError {
136
136
  constructor(status: number, body: ApiErrorBody);
137
137
  }
138
+ /** The code an import answers with when some rows were understood and then could not be stored. */
139
+ export declare const IMPORT_INCOMPLETE_CODE = "IMPORT_INCOMPLETE";
140
+ /**
141
+ * An import that did not finish: some rows could not be stored, or some
142
+ * verdicts that arrived with them could not be written.
143
+ *
144
+ * The rows it names were read and understood, so there is nothing to fix in
145
+ * the file: this is a storage failure, not a shape one. The full outcome is on
146
+ * {@link LoopImportIncompleteError.outcome} — what landed, which conversations
147
+ * it became, and which row positions did not — because a failure that reports
148
+ * only "it failed" leaves the caller with no move except sending everything
149
+ * again.
150
+ *
151
+ * HOW TO RECOVER. Send the same rows again with `importId`, the token this
152
+ * call was filed under. That is what makes the second call a RETRY: rows that
153
+ * already arrived come back under `already_present` instead of being stored
154
+ * twice, and a verdict that failed to write is attempted again. Repeat it
155
+ * exactly; a call that repeats nothing is a second import, on purpose, because
156
+ * a repeat is otherwise indistinguishable from the next page of a longer file.
157
+ *
158
+ * THE WHOLE FILE, not the rows this names, unless you sent `row_ids`. Without
159
+ * them a row is identified by its place in the call, so a shorter list moves
160
+ * every row after the gap and each is imported again.
161
+ */
162
+ export declare class LoopImportIncompleteError extends ApiError {
163
+ /** Counts, notes and ids for the import, in the same shape a success returns. */
164
+ readonly outcome: LoopImportResult;
165
+ constructor(status: number, body: ApiErrorBody);
166
+ /**
167
+ * The token to repeat as `import_id` to retry this call. Without it the retry
168
+ * is a second import rather than a retry, and what already landed is stored
169
+ * again.
170
+ */
171
+ get importId(): string | undefined;
172
+ /**
173
+ * Row positions that were not stored, counting from 1 in the order you sent
174
+ * them. They say what is missing; they are a smaller file to send back only
175
+ * if you sent `row_ids`.
176
+ */
177
+ get notSavedRows(): number[];
178
+ /**
179
+ * Row positions whose conversation stored and whose verdict did not. Those
180
+ * conversations are in the loop waiting for review; retrying records the
181
+ * verdict.
182
+ */
183
+ get verdictsNotSavedRows(): number[];
184
+ }
138
185
  /** Whether a machine code is one of the permanent GPU rejections. */
139
186
  export declare function isPermanentGpuCode(code: string | undefined): boolean;
140
187
  /**
package/dist/client.js CHANGED
@@ -191,6 +191,63 @@ export class ComingSoonError extends ApiError {
191
191
  this.name = 'ComingSoonError';
192
192
  }
193
193
  }
194
+ /** The code an import answers with when some rows were understood and then could not be stored. */
195
+ export const IMPORT_INCOMPLETE_CODE = 'IMPORT_INCOMPLETE';
196
+ /**
197
+ * An import that did not finish: some rows could not be stored, or some
198
+ * verdicts that arrived with them could not be written.
199
+ *
200
+ * The rows it names were read and understood, so there is nothing to fix in
201
+ * the file: this is a storage failure, not a shape one. The full outcome is on
202
+ * {@link LoopImportIncompleteError.outcome} — what landed, which conversations
203
+ * it became, and which row positions did not — because a failure that reports
204
+ * only "it failed" leaves the caller with no move except sending everything
205
+ * again.
206
+ *
207
+ * HOW TO RECOVER. Send the same rows again with `importId`, the token this
208
+ * call was filed under. That is what makes the second call a RETRY: rows that
209
+ * already arrived come back under `already_present` instead of being stored
210
+ * twice, and a verdict that failed to write is attempted again. Repeat it
211
+ * exactly; a call that repeats nothing is a second import, on purpose, because
212
+ * a repeat is otherwise indistinguishable from the next page of a longer file.
213
+ *
214
+ * THE WHOLE FILE, not the rows this names, unless you sent `row_ids`. Without
215
+ * them a row is identified by its place in the call, so a shorter list moves
216
+ * every row after the gap and each is imported again.
217
+ */
218
+ export class LoopImportIncompleteError extends ApiError {
219
+ /** Counts, notes and ids for the import, in the same shape a success returns. */
220
+ outcome;
221
+ constructor(status, body) {
222
+ super(status, body);
223
+ this.name = 'LoopImportIncompleteError';
224
+ this.outcome = body;
225
+ }
226
+ /**
227
+ * The token to repeat as `import_id` to retry this call. Without it the retry
228
+ * is a second import rather than a retry, and what already landed is stored
229
+ * again.
230
+ */
231
+ get importId() {
232
+ return this.outcome.import_id;
233
+ }
234
+ /**
235
+ * Row positions that were not stored, counting from 1 in the order you sent
236
+ * them. They say what is missing; they are a smaller file to send back only
237
+ * if you sent `row_ids`.
238
+ */
239
+ get notSavedRows() {
240
+ return this.outcome.not_saved_rows ?? [];
241
+ }
242
+ /**
243
+ * Row positions whose conversation stored and whose verdict did not. Those
244
+ * conversations are in the loop waiting for review; retrying records the
245
+ * verdict.
246
+ */
247
+ get verdictsNotSavedRows() {
248
+ return this.outcome.verdicts_not_saved_rows ?? [];
249
+ }
250
+ }
194
251
  /** Whether a machine code is one of the permanent GPU rejections. */
195
252
  export function isPermanentGpuCode(code) {
196
253
  return typeof code === 'string' && PERMANENT_GPU_CODES.includes(code);
@@ -225,6 +282,12 @@ export function buildApiError(status, body) {
225
282
  if (code && COMING_SOON_CODES.includes(code)) {
226
283
  return new ComingSoonError(status, body);
227
284
  }
285
+ // A half-finished import carries its whole outcome in the body. Leaving it as
286
+ // a plain ApiError makes the counts reachable only by casting `.body`, which
287
+ // is the same as not publishing them.
288
+ if (code === IMPORT_INCOMPLETE_CODE) {
289
+ return new LoopImportIncompleteError(status, body);
290
+ }
228
291
  return new ApiError(status, body);
229
292
  }
230
293
  // ============================================================================
package/dist/index.d.ts CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export declare const VERSION = "0.2.1-rc.143";
39
+ export declare const VERSION = "0.2.1-rc.151";
40
40
  export declare class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  readonly models: Models;
@@ -67,7 +67,7 @@ export declare class RunBiOS {
67
67
  }
68
68
  /** @deprecated Use {@link RunBiOS}. */
69
69
  export { RunBiOS as BiOS };
70
- export { ApiError, GpuRejectionError, CapacityUnavailableError, ComingSoonError, CAPACITY_UNAVAILABLE_CODE, COMING_SOON_CODES, PERMANENT_GPU_CODES, isPermanentGpuCode, gpuRejectionCodeForReason, } from './client.js';
70
+ export { ApiError, GpuRejectionError, CapacityUnavailableError, ComingSoonError, LoopImportIncompleteError, CAPACITY_UNAVAILABLE_CODE, COMING_SOON_CODES, IMPORT_INCOMPLETE_CODE, PERMANENT_GPU_CODES, isPermanentGpuCode, gpuRejectionCodeForReason, } from './client.js';
71
71
  export { Models } from './resources/models.js';
72
72
  export { Datasets } from './resources/datasets.js';
73
73
  export { Integrations, type HuggingFaceIntegration, type IntegrationBrowseParams } from './resources/integrations.js';
package/dist/index.js CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export const VERSION = '0.2.1-rc.143';
39
+ export const VERSION = '0.2.1-rc.151';
40
40
  export class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  models;
@@ -95,7 +95,7 @@ export { RunBiOS as BiOS };
95
95
  // ---------------------------------------------------------------------------
96
96
  // Re-exports
97
97
  // ---------------------------------------------------------------------------
98
- export { ApiError, GpuRejectionError, CapacityUnavailableError, ComingSoonError, CAPACITY_UNAVAILABLE_CODE, COMING_SOON_CODES, PERMANENT_GPU_CODES, isPermanentGpuCode, gpuRejectionCodeForReason, } from './client.js';
98
+ export { ApiError, GpuRejectionError, CapacityUnavailableError, ComingSoonError, LoopImportIncompleteError, CAPACITY_UNAVAILABLE_CODE, COMING_SOON_CODES, IMPORT_INCOMPLETE_CODE, PERMANENT_GPU_CODES, isPermanentGpuCode, gpuRejectionCodeForReason, } from './client.js';
99
99
  export { Models } from './resources/models.js';
100
100
  export { Datasets } from './resources/datasets.js';
101
101
  export { Integrations } from './resources/integrations.js';
@@ -84,6 +84,37 @@ export declare class Loop {
84
84
  * filed under: "everything in one bucket called import" is a corpus nobody
85
85
  * can slice afterwards. At most 5000 rows per call, because the call is
86
86
  * synchronous and somebody is waiting on it.
87
+ *
88
+ * WHAT COMES BACK. `imported` is what this call created and `trace_ids` names
89
+ * the conversations from the file that are now in the loop, so a caller can
90
+ * label, review or delete them. `by_shape`, `reviewed` and `needs_review`
91
+ * count what THIS CALL created, not what was merely recognised and not rows
92
+ * an earlier import already had, and the sentences in `notes` are written
93
+ * from those counts, so the numbers and the prose describe the same call and
94
+ * a re-send cannot report a reviewed conversation as waiting for review.
95
+ * `refused` is rows whose shape could not be read, which the file fixes;
96
+ * `not_saved` is rows that were understood and then could not be stored,
97
+ * which the file cannot fix.
98
+ *
99
+ * A call with any `not_saved`, or any `verdicts_not_saved` (the conversation
100
+ * stored and the judgement that came with it did not), REJECTS with
101
+ * `LoopImportIncompleteError`, whose `outcome` carries all of the above and
102
+ * whose `notSavedRows` and `verdictsNotSavedRows` name the positions.
103
+ *
104
+ * HOW A RETRY IS SAFE. Every response carries `import_id`, the token this
105
+ * call was filed under. Send the same rows again with it, and rows that
106
+ * already arrived come back on `already_present` instead of being stored
107
+ * twice, while a verdict that failed to write is attempted again. A call that
108
+ * repeats no token is its own import, deliberately: two calls carrying the
109
+ * same rows are as likely to be two pages of one export as one call twice,
110
+ * and guessing "retry" silently drops the second copy of every conversation a
111
+ * file lists more than once.
112
+ *
113
+ * SENDING A FILE IN PAGES. Either give each row its own `row_ids` entry, and
114
+ * then chunking, ordering and subsets all stop mattering; or send one
115
+ * `import_id` with the `row_offset` each page starts at. Do not send back
116
+ * only the rows `notSavedRows` names unless you used `row_ids`: without them
117
+ * a shorter list moves every row after the gap and imports it again.
87
118
  */
88
119
  importRows(params: LoopImportParams): Promise<LoopImportResult>;
89
120
  /** List captured conversations, newest first. */
@@ -212,7 +243,22 @@ export declare class Loop {
212
243
  key: string;
213
244
  label: string;
214
245
  }>;
215
- /** Every label in the workspace, with how many conversations carry it. */
246
+ /**
247
+ * Every label in the workspace, with how many conversations carry it and
248
+ * which master groups those conversations are in.
249
+ *
250
+ * TWO FIELDS ANSWER THE GROUP QUESTION, AND THEY ANSWER DIFFERENT ONES.
251
+ * `parent` is the group ALL of a label's conversations are in, and it is
252
+ * null the moment they disagree. `parents` is every group ANY of them are
253
+ * in, with how many of them are in each.
254
+ *
255
+ * Build a list of master groups out of `parents`. A label with 12
256
+ * conversations, 5 of them under `support`, reports `parent: null` and
257
+ * `parents: [{parent: 'support', traces: 5}]`, so code that reads only
258
+ * `parent` sees no group at all for it. The registry used to answer
259
+ * `parent: "support"` for all twelve, which named the group but put seven
260
+ * conversations in it that nobody had put there.
261
+ */
216
262
  listLabels(): Promise<LoopLabelCount[]>;
217
263
  /**
218
264
  * Build a training set from the feedback recorded so far.
@@ -422,11 +468,34 @@ export declare class Loop {
422
468
  * scoring something the rubric never asked for, is refused and reported in
423
469
  * `rejected` rather than silently dropped.
424
470
  *
425
- * Pass `finish` to close a run you have decided to stop early. Without it an
426
- * unfinished run stays open, which is the honest state for work that was
427
- * abandoned rather than completed.
471
+ * Pass `finish` to close the run in the same call once you have nothing left
472
+ * to send. To close a run WITHOUT scores, call `stopRun`: an empty verdict
473
+ * list with `finish` does the same thing and reads like a mistake.
428
474
  */
429
475
  postVerdicts(runId: string, verdicts: LoopJudgeVerdict[], finish?: boolean): Promise<LoopJudgeVerdictResult>;
476
+ /**
477
+ * Close a run that has not finished.
478
+ *
479
+ * A judge may have one open run of its own at a time, so an open run BLOCKS
480
+ * the next one, and the runs that most need closing are the ones nobody can
481
+ * wait out: a run the agent has parked on an empty balance or a refused key
482
+ * stays open until the reason is fixed or somebody stops it.
483
+ *
484
+ * Nothing is deleted. Every verdict already recorded stays recorded, the
485
+ * counters keep saying how much of the selection was covered, and the
486
+ * conversations the run was holding are free for the next one. The run ends
487
+ * as `stopped` rather than `done`, so an interrupted pass and a completed
488
+ * one do not read alike.
489
+ *
490
+ * A run that finished on its own is not rewritten: stopping one answers 409.
491
+ *
492
+ * Returns the stopped run itself, like `startRun` and `getRun`, not the
493
+ * `{run}` envelope the service sends. Every other single-run method in this
494
+ * class unwraps, and `Loop.stop_run` in the Python SDK does too, so a
495
+ * caller who wrote `(await loop.stopRun(id)).status` against one of its
496
+ * siblings is right here as well.
497
+ */
498
+ stopRun(runId: string): Promise<LoopJudgeRun>;
430
499
  /**
431
500
  * Can an agent run here, is one running, and is it on for this workspace.
432
501
  * `available` is about the environment; `online` about the worker;
@@ -494,6 +563,14 @@ export declare class Loop {
494
563
  *
495
564
  * Send the `terms_version` it returns back in `createTrainingRule`. Read the
496
565
  * figures out of this response rather than inventing ceilings of your own.
566
+ *
567
+ * `valid: false` with an EMPTY `refusals` list is not nothing: check
568
+ * `unreachable`, which names every peer the platform could not reach. The
569
+ * rule is savable in that state and would be paused, but the estimate around
570
+ * it is not trustworthy -- both `worst_hourly_*` are `0` -- so do not quote
571
+ * those figures to anyone. `model_revision` is NOT a signal here: a degraded
572
+ * pass echoes back whatever revision the request carried, so test
573
+ * `unreachable.length > 0` rather than the revision being empty.
497
574
  */
498
575
  preflightTrainingRule(params: TrainingRulePreflightRequest): Promise<TrainingRulePreflight>;
499
576
  /**
@@ -87,6 +87,37 @@ export class Loop {
87
87
  * filed under: "everything in one bucket called import" is a corpus nobody
88
88
  * can slice afterwards. At most 5000 rows per call, because the call is
89
89
  * synchronous and somebody is waiting on it.
90
+ *
91
+ * WHAT COMES BACK. `imported` is what this call created and `trace_ids` names
92
+ * the conversations from the file that are now in the loop, so a caller can
93
+ * label, review or delete them. `by_shape`, `reviewed` and `needs_review`
94
+ * count what THIS CALL created, not what was merely recognised and not rows
95
+ * an earlier import already had, and the sentences in `notes` are written
96
+ * from those counts, so the numbers and the prose describe the same call and
97
+ * a re-send cannot report a reviewed conversation as waiting for review.
98
+ * `refused` is rows whose shape could not be read, which the file fixes;
99
+ * `not_saved` is rows that were understood and then could not be stored,
100
+ * which the file cannot fix.
101
+ *
102
+ * A call with any `not_saved`, or any `verdicts_not_saved` (the conversation
103
+ * stored and the judgement that came with it did not), REJECTS with
104
+ * `LoopImportIncompleteError`, whose `outcome` carries all of the above and
105
+ * whose `notSavedRows` and `verdictsNotSavedRows` name the positions.
106
+ *
107
+ * HOW A RETRY IS SAFE. Every response carries `import_id`, the token this
108
+ * call was filed under. Send the same rows again with it, and rows that
109
+ * already arrived come back on `already_present` instead of being stored
110
+ * twice, while a verdict that failed to write is attempted again. A call that
111
+ * repeats no token is its own import, deliberately: two calls carrying the
112
+ * same rows are as likely to be two pages of one export as one call twice,
113
+ * and guessing "retry" silently drops the second copy of every conversation a
114
+ * file lists more than once.
115
+ *
116
+ * SENDING A FILE IN PAGES. Either give each row its own `row_ids` entry, and
117
+ * then chunking, ordering and subsets all stop mattering; or send one
118
+ * `import_id` with the `row_offset` each page starts at. Do not send back
119
+ * only the rows `notSavedRows` names unless you used `row_ids`: without them
120
+ * a shorter list moves every row after the gap and imports it again.
90
121
  */
91
122
  async importRows(params) {
92
123
  return this._http.fetchPost('/api/loop/import', params);
@@ -280,7 +311,22 @@ export class Loop {
280
311
  const q = key ? `?key=${encodeURIComponent(key)}` : '';
281
312
  return this._http.fetchDelete(`/api/loop/traces/${encodeURIComponent(traceId)}/labels/${encodeURIComponent(label)}${q}`);
282
313
  }
283
- /** Every label in the workspace, with how many conversations carry it. */
314
+ /**
315
+ * Every label in the workspace, with how many conversations carry it and
316
+ * which master groups those conversations are in.
317
+ *
318
+ * TWO FIELDS ANSWER THE GROUP QUESTION, AND THEY ANSWER DIFFERENT ONES.
319
+ * `parent` is the group ALL of a label's conversations are in, and it is
320
+ * null the moment they disagree. `parents` is every group ANY of them are
321
+ * in, with how many of them are in each.
322
+ *
323
+ * Build a list of master groups out of `parents`. A label with 12
324
+ * conversations, 5 of them under `support`, reports `parent: null` and
325
+ * `parents: [{parent: 'support', traces: 5}]`, so code that reads only
326
+ * `parent` sees no group at all for it. The registry used to answer
327
+ * `parent: "support"` for all twelve, which named the group but put seven
328
+ * conversations in it that nobody had put there.
329
+ */
284
330
  async listLabels() {
285
331
  const res = await this._http.fetchGet('/api/loop/labels');
286
332
  return res.labels;
@@ -589,13 +635,39 @@ export class Loop {
589
635
  * scoring something the rubric never asked for, is refused and reported in
590
636
  * `rejected` rather than silently dropped.
591
637
  *
592
- * Pass `finish` to close a run you have decided to stop early. Without it an
593
- * unfinished run stays open, which is the honest state for work that was
594
- * abandoned rather than completed.
638
+ * Pass `finish` to close the run in the same call once you have nothing left
639
+ * to send. To close a run WITHOUT scores, call `stopRun`: an empty verdict
640
+ * list with `finish` does the same thing and reads like a mistake.
595
641
  */
596
642
  async postVerdicts(runId, verdicts, finish = false) {
597
643
  return this._http.fetchPost(`/api/loop/runs/${encodeURIComponent(runId)}/verdicts`, { verdicts, finish });
598
644
  }
645
+ /**
646
+ * Close a run that has not finished.
647
+ *
648
+ * A judge may have one open run of its own at a time, so an open run BLOCKS
649
+ * the next one, and the runs that most need closing are the ones nobody can
650
+ * wait out: a run the agent has parked on an empty balance or a refused key
651
+ * stays open until the reason is fixed or somebody stops it.
652
+ *
653
+ * Nothing is deleted. Every verdict already recorded stays recorded, the
654
+ * counters keep saying how much of the selection was covered, and the
655
+ * conversations the run was holding are free for the next one. The run ends
656
+ * as `stopped` rather than `done`, so an interrupted pass and a completed
657
+ * one do not read alike.
658
+ *
659
+ * A run that finished on its own is not rewritten: stopping one answers 409.
660
+ *
661
+ * Returns the stopped run itself, like `startRun` and `getRun`, not the
662
+ * `{run}` envelope the service sends. Every other single-run method in this
663
+ * class unwraps, and `Loop.stop_run` in the Python SDK does too, so a
664
+ * caller who wrote `(await loop.stopRun(id)).status` against one of its
665
+ * siblings is right here as well.
666
+ */
667
+ async stopRun(runId) {
668
+ const res = await this._http.fetchPost(`/api/loop/runs/${encodeURIComponent(runId)}/stop`, {});
669
+ return res.run;
670
+ }
599
671
  // ── the agent ─────────────────────────────────────────────────────────
600
672
  //
601
673
  // The worker that calls a model on the workspace's behalf: it applies
@@ -700,6 +772,14 @@ export class Loop {
700
772
  *
701
773
  * Send the `terms_version` it returns back in `createTrainingRule`. Read the
702
774
  * figures out of this response rather than inventing ceilings of your own.
775
+ *
776
+ * `valid: false` with an EMPTY `refusals` list is not nothing: check
777
+ * `unreachable`, which names every peer the platform could not reach. The
778
+ * rule is savable in that state and would be paused, but the estimate around
779
+ * it is not trustworthy -- both `worst_hourly_*` are `0` -- so do not quote
780
+ * those figures to anyone. `model_revision` is NOT a signal here: a degraded
781
+ * pass echoes back whatever revision the request carried, so test
782
+ * `unreachable.length > 0` rather than the revision being empty.
703
783
  */
704
784
  async preflightTrainingRule(params) {
705
785
  return this._http.fetchPost('/api/loop/training-rules/preflight', params);
package/dist/types.d.ts CHANGED
@@ -2233,13 +2233,91 @@ export interface LoopImportParams {
2233
2233
  /** Applied to every row: importing a dump is when somebody knows what it is. */
2234
2234
  labels?: string[];
2235
2235
  attributes?: Record<string, string>;
2236
+ /**
2237
+ * Repeat the `import_id` a previous call reported to make this call a RETRY
2238
+ * of that one: rows that already arrived come back under `already_present`
2239
+ * instead of being stored again.
2240
+ *
2241
+ * Leave it out and this call is its own import. The endpoint will NOT guess:
2242
+ * two calls carrying the same rows are as likely to be two pages of one
2243
+ * export as one call sent twice, and guessing "retry" silently drops the
2244
+ * second copy of every conversation a file lists more than once. Every
2245
+ * response carries the token it was filed under, so a retry is always
2246
+ * available and never has to be inferred.
2247
+ *
2248
+ * SEND THE ROWS AS YOU SENT THEM. Without `row_ids` a row is matched on its
2249
+ * text, and the match survives re-ordered keys and different whitespace but
2250
+ * NOT a number that has been re-spelled: `100`, `1e2` and `100.0` are three
2251
+ * different rows. A round trip through most JSON libraries re-spells numbers
2252
+ * (Python turns `1e2` into `100.0`), so a retry built by re-serialising a
2253
+ * parsed file can store rows a second time. Keep the bytes you sent, or send
2254
+ * `row_ids` and stop depending on the text at all.
2255
+ */
2256
+ import_id?: string;
2257
+ /**
2258
+ * Where this page starts in the file, counting from 0. Send it when you split
2259
+ * one import across several calls under one `import_id`, so two pages are not
2260
+ * matched against each other row for row. Ignored when you send `row_ids`.
2261
+ */
2262
+ row_offset?: number;
2263
+ /**
2264
+ * Your own id for each row, in the same order as `rows`: a ticket number, a
2265
+ * conversation id, whatever the export already carries.
2266
+ *
2267
+ * The strongest form of identity, and the one to reach for with a file big
2268
+ * enough to page. It SUPERSEDES `import_id` and `row_offset`: with it,
2269
+ * chunking, ordering and subsets all stop mattering and any part of the file
2270
+ * can be re-sent exactly. One per row or none at all, and no two rows in a
2271
+ * call may share an id: both are refused rather than silently merging two
2272
+ * conversations into one.
2273
+ *
2274
+ * THE ID IS THE IDENTITY, AND THE ROW'S TEXT IS NOT PART OF IT. A row sent
2275
+ * again under an id this source already imported comes back under
2276
+ * `already_present` and the stored conversation is left as it was, even if
2277
+ * you changed its text. A CORRECTION DOES NOT LAND THIS WAY: send it under a
2278
+ * new id. That is the trade for making every retry, page and subset safe -
2279
+ * the text is never compared, so nothing your serialiser does to it can turn
2280
+ * a retry into a second import.
2281
+ */
2282
+ row_ids?: string[];
2236
2283
  }
2284
+ /**
2285
+ * What an import did.
2286
+ *
2287
+ * A call where every row stored resolves with this. A call where SOME rows
2288
+ * could not be stored rejects with {@link LoopImportIncompleteError}, whose
2289
+ * `outcome` is this same shape: the counts are the whole answer either way, so
2290
+ * read them from the error exactly as you would from a success.
2291
+ */
2237
2292
  export interface LoopImportResult {
2293
+ /**
2294
+ * The token this call was filed under, whether you sent one or it was minted
2295
+ * for you. Send the same rows again with this `import_id` to retry the call.
2296
+ * Present on every response, success or failure, because a call can succeed
2297
+ * and still need retrying when the answer never reached you.
2298
+ */
2299
+ import_id?: string;
2300
+ /** How many conversations this call created. */
2238
2301
  imported: number;
2239
2302
  /**
2240
- * How many arrived carrying a verdict. Reported apart from `imported`
2241
- * because it is the difference between data you can train on and data
2242
- * somebody still has to look at.
2303
+ * How many rows an earlier call under the SAME `import_id` (or carrying the
2304
+ * same `row_ids`) had already imported, and so were not stored a second time.
2305
+ * Zero on a first import, and zero on any call that repeated neither: a call
2306
+ * that does not say it is a retry is its own import, which is what lets a
2307
+ * file sent in pages land complete.
2308
+ */
2309
+ already_present?: number;
2310
+ /**
2311
+ * How many verdicts THIS CALL WROTE. Reported apart from `imported` because
2312
+ * it is the difference between data you can train on and data somebody still
2313
+ * has to look at.
2314
+ *
2315
+ * Re-sending a row whose verdict already landed writes nothing and counts
2316
+ * nothing, so a plain retry reads 0. Re-sending one that came back in
2317
+ * `verdicts_not_saved` DOES count here when the write succeeds this time,
2318
+ * even though the row itself is counted under `already_present`: that is the
2319
+ * first time anybody recorded that verdict, not a second reviewer. This is
2320
+ * the field to read to confirm a repair landed.
2243
2321
  */
2244
2322
  reviewed: number;
2245
2323
  needs_review: number;
@@ -2251,17 +2329,47 @@ export interface LoopImportResult {
2251
2329
  /**
2252
2330
  * How many rows were understood and then could not be stored. Counted apart
2253
2331
  * from `refused`, because the two are different claims and rewriting a file
2254
- * that was already correct fixes nothing. A call with any of these fails
2255
- * rather than succeeding, and the failure carries these same fields.
2332
+ * that was already correct fixes nothing. A call with any of these rejects
2333
+ * rather than resolving; read it off {@link LoopImportIncompleteError.outcome}.
2256
2334
  */
2257
2335
  not_saved?: number;
2258
- /** How many rows of each recognised shape, so a mis-shaped file is visible. */
2336
+ /**
2337
+ * WHICH rows those were, counting from 1 in the order you sent them, so you
2338
+ * can say what is missing rather than only how much.
2339
+ *
2340
+ * NOT A SMALLER FILE TO SEND BACK unless you sent `row_ids`. Without them a
2341
+ * row is identified by its place in the call, so a shorter list moves every
2342
+ * row after the gap and each one is stored again. Re-send the whole set of
2343
+ * rows, in the same order, with the same `import_id`; what already arrived
2344
+ * comes back under `already_present`. With `row_ids` a row carries its own
2345
+ * identity and any subset is exact.
2346
+ */
2347
+ not_saved_rows?: number[];
2348
+ /**
2349
+ * Rows whose CONVERSATION stored and whose VERDICT did not, and which rows
2350
+ * those were. A preference pair or a thumbs label is two writes, and the
2351
+ * second can fail on its own: the conversation is then in the loop carrying
2352
+ * nobody's judgement, counted under `needs_review` rather than `reviewed`,
2353
+ * and not trainable. The call rejects with {@link LoopImportIncompleteError},
2354
+ * and sending the rows again under the same `import_id` records the verdict
2355
+ * without storing the conversation twice.
2356
+ */
2357
+ verdicts_not_saved?: number;
2358
+ verdicts_not_saved_rows?: number[];
2359
+ /**
2360
+ * How many conversations of each shape THIS CALL created. Rows that were
2361
+ * refused, rows that failed to store and rows an earlier import already had
2362
+ * are not in here: the counts and the sentences in `notes` describe the same
2363
+ * call, so neither can claim a conversation is waiting for review when none
2364
+ * was written.
2365
+ */
2259
2366
  by_shape: Record<string, number>;
2260
2367
  refused_why: Record<string, number>;
2261
2368
  /**
2262
- * The conversations this call created, in file order. Without them a caller
2263
- * whose import half-failed has a number and no way to reach what landed: no
2264
- * way to label it, review it, or delete it and start again.
2369
+ * The conversations from this file that are now in the loop, in file order:
2370
+ * the ones this call created and the ones it found already there. Without
2371
+ * them a caller whose import half-failed has a number and no way to reach
2372
+ * what landed: no way to label it, review it, or delete it and start again.
2265
2373
  */
2266
2374
  trace_ids?: string[];
2267
2375
  notes: string[];
@@ -2350,7 +2458,11 @@ export interface LoopDatasetCreateParams {
2350
2458
  to?: string;
2351
2459
  /** Narrow to one slice. A label with children selects them too. */
2352
2460
  label?: string;
2353
- /** Take this many at random from what the filters matched. */
2461
+ /**
2462
+ * Take this many of what the filters matched. Which ones is arbitrary but
2463
+ * REPEATABLE: the same conversations always give the same slice, so two
2464
+ * builds of one spec describe the same set.
2465
+ */
2354
2466
  sample?: number;
2355
2467
  /** Holds back your most recent work, not a random slice. 0-50. */
2356
2468
  holdout_percent?: number;
@@ -2535,8 +2647,13 @@ export interface LoopJudgeRun {
2535
2647
  * it for a week and items were still waiting. Nothing is deleted and posting
2536
2648
  * verdicts to it still works and still closes it as done. It exists so that
2537
2649
  * `open` keeps meaning "somebody is working on this".
2650
+ *
2651
+ * `stopped` is a run somebody closed on purpose before it finished, with
2652
+ * `stopRun`. Separate from `done` because a run that covered three of forty
2653
+ * conversations did not finish its work: every score it recorded is kept
2654
+ * either way, and reading one as the other overstates what was evaluated.
2538
2655
  */
2539
- status: 'open' | 'done' | 'abandoned';
2656
+ status: 'open' | 'done' | 'abandoned' | 'stopped';
2540
2657
  instructions: string;
2541
2658
  dimensions: LoopJudgeDimension[];
2542
2659
  model: string | null;
@@ -2817,9 +2934,34 @@ export interface LoopLabel {
2817
2934
  /** The dimension this value belongs to. `tag` for a bare name. */
2818
2935
  key: string;
2819
2936
  }
2937
+ /** One master group, and how many of a label's conversations are in it. */
2938
+ export interface LoopLabelParentCount {
2939
+ parent: string;
2940
+ traces: number;
2941
+ }
2820
2942
  export interface LoopLabelCount {
2821
2943
  label: string;
2944
+ /**
2945
+ * The master group ALL of this label's conversations are in, and nothing
2946
+ * else. Null when they disagree: the registry used to answer the
2947
+ * alphabetically last parent over a mixed group, so `billing` was reported
2948
+ * under `support` while seven of its twelve conversations were in no group
2949
+ * at all. Render "(in x)" from this field only.
2950
+ */
2822
2951
  parent: string | null;
2952
+ /**
2953
+ * Every master group ANY of them are in, sorted, with how many of this
2954
+ * label's conversations are in each. Empty when there are none.
2955
+ *
2956
+ * Read this, not `parent`, to discover which master groups exist: a label
2957
+ * whose conversations disagree still belongs partly to a real group, and
2958
+ * `parent` is null for it, so a picker built from `parent` alone loses the
2959
+ * group along with the false claim. The count is per group rather than the
2960
+ * label's own total, because adding a label's whole count to its parent is
2961
+ * the same mistake one level down. What is in no group at all is `traces`
2962
+ * minus the sum of these.
2963
+ */
2964
+ parents: LoopLabelParentCount[];
2823
2965
  traces: number;
2824
2966
  key: string;
2825
2967
  }
@@ -3002,6 +3144,22 @@ export interface TrainingRulePreflight {
3002
3144
  eval_ceiling_cents: number;
3003
3145
  warnings: string[];
3004
3146
  refusals: TrainingRulePreflightRefusal[];
3147
+ /**
3148
+ * Every peer the platform could not reach on this pass, one entry each, in
3149
+ * the same `{stage, code, message}` shape as a refusal.
3150
+ *
3151
+ * A warning, not a refusal: an unreachable peer judged nothing, so it never
3152
+ * refuses a save. It is still why the rule cannot fire — both
3153
+ * `worst_hourly_*` come back `0` — so `valid` is `false`. Key a "cannot
3154
+ * save" state on this when `refusals` is empty; the same sentences are also
3155
+ * in `warnings`, so render one or the other.
3156
+ *
3157
+ * Key it on THIS, not on `model_revision`. A degraded pass echoes back the
3158
+ * `model_revision` the request carried rather than emptying it, so
3159
+ * `if (!model_revision)` is false on exactly the passes it was meant to
3160
+ * catch. Use `unreachable.length > 0`, or `/^[0-9a-f]{40}$/`.
3161
+ */
3162
+ unreachable: TrainingRulePreflightRefusal[];
3005
3163
  key_check: TrainingRuleKeyCheck;
3006
3164
  terms_text: string;
3007
3165
  terms_version: string;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.1-rc.143",
3
+ "version": "0.2.1-rc.151",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",