runbios-sdk 0.2.1-rc.143 → 0.2.1-rc.151
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/client.d.ts +48 -1
- package/dist/client.js +63 -0
- package/dist/index.d.ts +2 -2
- package/dist/index.js +2 -2
- package/dist/resources/loop.d.ts +81 -4
- package/dist/resources/loop.js +84 -4
- package/dist/types.d.ts +169 -11
- package/package.json +1 -1
package/dist/client.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import type { BiOSConfig, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement } from './types.js';
|
|
1
|
+
import type { BiOSConfig, ApiErrorBody, AvailableGpuAlternative, CapacityMinimumRequirement, LoopImportResult } from './types.js';
|
|
2
2
|
/**
|
|
3
3
|
* Typed error thrown by every SDK method when the API returns a non-2xx status.
|
|
4
4
|
*
|
|
@@ -135,6 +135,53 @@ export declare const COMING_SOON_CODES: readonly ["TRAINING_COMING_SOON", "DATAS
|
|
|
135
135
|
export declare class ComingSoonError extends ApiError {
|
|
136
136
|
constructor(status: number, body: ApiErrorBody);
|
|
137
137
|
}
|
|
138
|
+
/** The code an import answers with when some rows were understood and then could not be stored. */
|
|
139
|
+
export declare const IMPORT_INCOMPLETE_CODE = "IMPORT_INCOMPLETE";
|
|
140
|
+
/**
|
|
141
|
+
* An import that did not finish: some rows could not be stored, or some
|
|
142
|
+
* verdicts that arrived with them could not be written.
|
|
143
|
+
*
|
|
144
|
+
* The rows it names were read and understood, so there is nothing to fix in
|
|
145
|
+
* the file: this is a storage failure, not a shape one. The full outcome is on
|
|
146
|
+
* {@link LoopImportIncompleteError.outcome} — what landed, which conversations
|
|
147
|
+
* it became, and which row positions did not — because a failure that reports
|
|
148
|
+
* only "it failed" leaves the caller with no move except sending everything
|
|
149
|
+
* again.
|
|
150
|
+
*
|
|
151
|
+
* HOW TO RECOVER. Send the same rows again with `importId`, the token this
|
|
152
|
+
* call was filed under. That is what makes the second call a RETRY: rows that
|
|
153
|
+
* already arrived come back under `already_present` instead of being stored
|
|
154
|
+
* twice, and a verdict that failed to write is attempted again. Repeat it
|
|
155
|
+
* exactly; a call that repeats nothing is a second import, on purpose, because
|
|
156
|
+
* a repeat is otherwise indistinguishable from the next page of a longer file.
|
|
157
|
+
*
|
|
158
|
+
* THE WHOLE FILE, not the rows this names, unless you sent `row_ids`. Without
|
|
159
|
+
* them a row is identified by its place in the call, so a shorter list moves
|
|
160
|
+
* every row after the gap and each is imported again.
|
|
161
|
+
*/
|
|
162
|
+
export declare class LoopImportIncompleteError extends ApiError {
|
|
163
|
+
/** Counts, notes and ids for the import, in the same shape a success returns. */
|
|
164
|
+
readonly outcome: LoopImportResult;
|
|
165
|
+
constructor(status: number, body: ApiErrorBody);
|
|
166
|
+
/**
|
|
167
|
+
* The token to repeat as `import_id` to retry this call. Without it the retry
|
|
168
|
+
* is a second import rather than a retry, and what already landed is stored
|
|
169
|
+
* again.
|
|
170
|
+
*/
|
|
171
|
+
get importId(): string | undefined;
|
|
172
|
+
/**
|
|
173
|
+
* Row positions that were not stored, counting from 1 in the order you sent
|
|
174
|
+
* them. They say what is missing; they are a smaller file to send back only
|
|
175
|
+
* if you sent `row_ids`.
|
|
176
|
+
*/
|
|
177
|
+
get notSavedRows(): number[];
|
|
178
|
+
/**
|
|
179
|
+
* Row positions whose conversation stored and whose verdict did not. Those
|
|
180
|
+
* conversations are in the loop waiting for review; retrying records the
|
|
181
|
+
* verdict.
|
|
182
|
+
*/
|
|
183
|
+
get verdictsNotSavedRows(): number[];
|
|
184
|
+
}
|
|
138
185
|
/** Whether a machine code is one of the permanent GPU rejections. */
|
|
139
186
|
export declare function isPermanentGpuCode(code: string | undefined): boolean;
|
|
140
187
|
/**
|
package/dist/client.js
CHANGED
|
@@ -191,6 +191,63 @@ export class ComingSoonError extends ApiError {
|
|
|
191
191
|
this.name = 'ComingSoonError';
|
|
192
192
|
}
|
|
193
193
|
}
|
|
194
|
+
/** The code an import answers with when some rows were understood and then could not be stored. */
|
|
195
|
+
export const IMPORT_INCOMPLETE_CODE = 'IMPORT_INCOMPLETE';
|
|
196
|
+
/**
|
|
197
|
+
* An import that did not finish: some rows could not be stored, or some
|
|
198
|
+
* verdicts that arrived with them could not be written.
|
|
199
|
+
*
|
|
200
|
+
* The rows it names were read and understood, so there is nothing to fix in
|
|
201
|
+
* the file: this is a storage failure, not a shape one. The full outcome is on
|
|
202
|
+
* {@link LoopImportIncompleteError.outcome} — what landed, which conversations
|
|
203
|
+
* it became, and which row positions did not — because a failure that reports
|
|
204
|
+
* only "it failed" leaves the caller with no move except sending everything
|
|
205
|
+
* again.
|
|
206
|
+
*
|
|
207
|
+
* HOW TO RECOVER. Send the same rows again with `importId`, the token this
|
|
208
|
+
* call was filed under. That is what makes the second call a RETRY: rows that
|
|
209
|
+
* already arrived come back under `already_present` instead of being stored
|
|
210
|
+
* twice, and a verdict that failed to write is attempted again. Repeat it
|
|
211
|
+
* exactly; a call that repeats nothing is a second import, on purpose, because
|
|
212
|
+
* a repeat is otherwise indistinguishable from the next page of a longer file.
|
|
213
|
+
*
|
|
214
|
+
* THE WHOLE FILE, not the rows this names, unless you sent `row_ids`. Without
|
|
215
|
+
* them a row is identified by its place in the call, so a shorter list moves
|
|
216
|
+
* every row after the gap and each is imported again.
|
|
217
|
+
*/
|
|
218
|
+
export class LoopImportIncompleteError extends ApiError {
|
|
219
|
+
/** Counts, notes and ids for the import, in the same shape a success returns. */
|
|
220
|
+
outcome;
|
|
221
|
+
constructor(status, body) {
|
|
222
|
+
super(status, body);
|
|
223
|
+
this.name = 'LoopImportIncompleteError';
|
|
224
|
+
this.outcome = body;
|
|
225
|
+
}
|
|
226
|
+
/**
|
|
227
|
+
* The token to repeat as `import_id` to retry this call. Without it the retry
|
|
228
|
+
* is a second import rather than a retry, and what already landed is stored
|
|
229
|
+
* again.
|
|
230
|
+
*/
|
|
231
|
+
get importId() {
|
|
232
|
+
return this.outcome.import_id;
|
|
233
|
+
}
|
|
234
|
+
/**
|
|
235
|
+
* Row positions that were not stored, counting from 1 in the order you sent
|
|
236
|
+
* them. They say what is missing; they are a smaller file to send back only
|
|
237
|
+
* if you sent `row_ids`.
|
|
238
|
+
*/
|
|
239
|
+
get notSavedRows() {
|
|
240
|
+
return this.outcome.not_saved_rows ?? [];
|
|
241
|
+
}
|
|
242
|
+
/**
|
|
243
|
+
* Row positions whose conversation stored and whose verdict did not. Those
|
|
244
|
+
* conversations are in the loop waiting for review; retrying records the
|
|
245
|
+
* verdict.
|
|
246
|
+
*/
|
|
247
|
+
get verdictsNotSavedRows() {
|
|
248
|
+
return this.outcome.verdicts_not_saved_rows ?? [];
|
|
249
|
+
}
|
|
250
|
+
}
|
|
194
251
|
/** Whether a machine code is one of the permanent GPU rejections. */
|
|
195
252
|
export function isPermanentGpuCode(code) {
|
|
196
253
|
return typeof code === 'string' && PERMANENT_GPU_CODES.includes(code);
|
|
@@ -225,6 +282,12 @@ export function buildApiError(status, body) {
|
|
|
225
282
|
if (code && COMING_SOON_CODES.includes(code)) {
|
|
226
283
|
return new ComingSoonError(status, body);
|
|
227
284
|
}
|
|
285
|
+
// A half-finished import carries its whole outcome in the body. Leaving it as
|
|
286
|
+
// a plain ApiError makes the counts reachable only by casting `.body`, which
|
|
287
|
+
// is the same as not publishing them.
|
|
288
|
+
if (code === IMPORT_INCOMPLETE_CODE) {
|
|
289
|
+
return new LoopImportIncompleteError(status, body);
|
|
290
|
+
}
|
|
228
291
|
return new ApiError(status, body);
|
|
229
292
|
}
|
|
230
293
|
// ============================================================================
|
package/dist/index.d.ts
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export declare const VERSION = "0.2.1-rc.
|
|
39
|
+
export declare const VERSION = "0.2.1-rc.151";
|
|
40
40
|
export declare class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
readonly models: Models;
|
|
@@ -67,7 +67,7 @@ export declare class RunBiOS {
|
|
|
67
67
|
}
|
|
68
68
|
/** @deprecated Use {@link RunBiOS}. */
|
|
69
69
|
export { RunBiOS as BiOS };
|
|
70
|
-
export { ApiError, GpuRejectionError, CapacityUnavailableError, ComingSoonError, CAPACITY_UNAVAILABLE_CODE, COMING_SOON_CODES, PERMANENT_GPU_CODES, isPermanentGpuCode, gpuRejectionCodeForReason, } from './client.js';
|
|
70
|
+
export { ApiError, GpuRejectionError, CapacityUnavailableError, ComingSoonError, LoopImportIncompleteError, CAPACITY_UNAVAILABLE_CODE, COMING_SOON_CODES, IMPORT_INCOMPLETE_CODE, PERMANENT_GPU_CODES, isPermanentGpuCode, gpuRejectionCodeForReason, } from './client.js';
|
|
71
71
|
export { Models } from './resources/models.js';
|
|
72
72
|
export { Datasets } from './resources/datasets.js';
|
|
73
73
|
export { Integrations, type HuggingFaceIntegration, type IntegrationBrowseParams } from './resources/integrations.js';
|
package/dist/index.js
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export const VERSION = '0.2.1-rc.
|
|
39
|
+
export const VERSION = '0.2.1-rc.151';
|
|
40
40
|
export class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
models;
|
|
@@ -95,7 +95,7 @@ export { RunBiOS as BiOS };
|
|
|
95
95
|
// ---------------------------------------------------------------------------
|
|
96
96
|
// Re-exports
|
|
97
97
|
// ---------------------------------------------------------------------------
|
|
98
|
-
export { ApiError, GpuRejectionError, CapacityUnavailableError, ComingSoonError, CAPACITY_UNAVAILABLE_CODE, COMING_SOON_CODES, PERMANENT_GPU_CODES, isPermanentGpuCode, gpuRejectionCodeForReason, } from './client.js';
|
|
98
|
+
export { ApiError, GpuRejectionError, CapacityUnavailableError, ComingSoonError, LoopImportIncompleteError, CAPACITY_UNAVAILABLE_CODE, COMING_SOON_CODES, IMPORT_INCOMPLETE_CODE, PERMANENT_GPU_CODES, isPermanentGpuCode, gpuRejectionCodeForReason, } from './client.js';
|
|
99
99
|
export { Models } from './resources/models.js';
|
|
100
100
|
export { Datasets } from './resources/datasets.js';
|
|
101
101
|
export { Integrations } from './resources/integrations.js';
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -84,6 +84,37 @@ export declare class Loop {
|
|
|
84
84
|
* filed under: "everything in one bucket called import" is a corpus nobody
|
|
85
85
|
* can slice afterwards. At most 5000 rows per call, because the call is
|
|
86
86
|
* synchronous and somebody is waiting on it.
|
|
87
|
+
*
|
|
88
|
+
* WHAT COMES BACK. `imported` is what this call created and `trace_ids` names
|
|
89
|
+
* the conversations from the file that are now in the loop, so a caller can
|
|
90
|
+
* label, review or delete them. `by_shape`, `reviewed` and `needs_review`
|
|
91
|
+
* count what THIS CALL created, not what was merely recognised and not rows
|
|
92
|
+
* an earlier import already had, and the sentences in `notes` are written
|
|
93
|
+
* from those counts, so the numbers and the prose describe the same call and
|
|
94
|
+
* a re-send cannot report a reviewed conversation as waiting for review.
|
|
95
|
+
* `refused` is rows whose shape could not be read, which the file fixes;
|
|
96
|
+
* `not_saved` is rows that were understood and then could not be stored,
|
|
97
|
+
* which the file cannot fix.
|
|
98
|
+
*
|
|
99
|
+
* A call with any `not_saved`, or any `verdicts_not_saved` (the conversation
|
|
100
|
+
* stored and the judgement that came with it did not), REJECTS with
|
|
101
|
+
* `LoopImportIncompleteError`, whose `outcome` carries all of the above and
|
|
102
|
+
* whose `notSavedRows` and `verdictsNotSavedRows` name the positions.
|
|
103
|
+
*
|
|
104
|
+
* HOW A RETRY IS SAFE. Every response carries `import_id`, the token this
|
|
105
|
+
* call was filed under. Send the same rows again with it, and rows that
|
|
106
|
+
* already arrived come back on `already_present` instead of being stored
|
|
107
|
+
* twice, while a verdict that failed to write is attempted again. A call that
|
|
108
|
+
* repeats no token is its own import, deliberately: two calls carrying the
|
|
109
|
+
* same rows are as likely to be two pages of one export as one call twice,
|
|
110
|
+
* and guessing "retry" silently drops the second copy of every conversation a
|
|
111
|
+
* file lists more than once.
|
|
112
|
+
*
|
|
113
|
+
* SENDING A FILE IN PAGES. Either give each row its own `row_ids` entry, and
|
|
114
|
+
* then chunking, ordering and subsets all stop mattering; or send one
|
|
115
|
+
* `import_id` with the `row_offset` each page starts at. Do not send back
|
|
116
|
+
* only the rows `notSavedRows` names unless you used `row_ids`: without them
|
|
117
|
+
* a shorter list moves every row after the gap and imports it again.
|
|
87
118
|
*/
|
|
88
119
|
importRows(params: LoopImportParams): Promise<LoopImportResult>;
|
|
89
120
|
/** List captured conversations, newest first. */
|
|
@@ -212,7 +243,22 @@ export declare class Loop {
|
|
|
212
243
|
key: string;
|
|
213
244
|
label: string;
|
|
214
245
|
}>;
|
|
215
|
-
/**
|
|
246
|
+
/**
|
|
247
|
+
* Every label in the workspace, with how many conversations carry it and
|
|
248
|
+
* which master groups those conversations are in.
|
|
249
|
+
*
|
|
250
|
+
* TWO FIELDS ANSWER THE GROUP QUESTION, AND THEY ANSWER DIFFERENT ONES.
|
|
251
|
+
* `parent` is the group ALL of a label's conversations are in, and it is
|
|
252
|
+
* null the moment they disagree. `parents` is every group ANY of them are
|
|
253
|
+
* in, with how many of them are in each.
|
|
254
|
+
*
|
|
255
|
+
* Build a list of master groups out of `parents`. A label with 12
|
|
256
|
+
* conversations, 5 of them under `support`, reports `parent: null` and
|
|
257
|
+
* `parents: [{parent: 'support', traces: 5}]`, so code that reads only
|
|
258
|
+
* `parent` sees no group at all for it. The registry used to answer
|
|
259
|
+
* `parent: "support"` for all twelve, which named the group but put seven
|
|
260
|
+
* conversations in it that nobody had put there.
|
|
261
|
+
*/
|
|
216
262
|
listLabels(): Promise<LoopLabelCount[]>;
|
|
217
263
|
/**
|
|
218
264
|
* Build a training set from the feedback recorded so far.
|
|
@@ -422,11 +468,34 @@ export declare class Loop {
|
|
|
422
468
|
* scoring something the rubric never asked for, is refused and reported in
|
|
423
469
|
* `rejected` rather than silently dropped.
|
|
424
470
|
*
|
|
425
|
-
* Pass `finish` to close
|
|
426
|
-
*
|
|
427
|
-
*
|
|
471
|
+
* Pass `finish` to close the run in the same call once you have nothing left
|
|
472
|
+
* to send. To close a run WITHOUT scores, call `stopRun`: an empty verdict
|
|
473
|
+
* list with `finish` does the same thing and reads like a mistake.
|
|
428
474
|
*/
|
|
429
475
|
postVerdicts(runId: string, verdicts: LoopJudgeVerdict[], finish?: boolean): Promise<LoopJudgeVerdictResult>;
|
|
476
|
+
/**
|
|
477
|
+
* Close a run that has not finished.
|
|
478
|
+
*
|
|
479
|
+
* A judge may have one open run of its own at a time, so an open run BLOCKS
|
|
480
|
+
* the next one, and the runs that most need closing are the ones nobody can
|
|
481
|
+
* wait out: a run the agent has parked on an empty balance or a refused key
|
|
482
|
+
* stays open until the reason is fixed or somebody stops it.
|
|
483
|
+
*
|
|
484
|
+
* Nothing is deleted. Every verdict already recorded stays recorded, the
|
|
485
|
+
* counters keep saying how much of the selection was covered, and the
|
|
486
|
+
* conversations the run was holding are free for the next one. The run ends
|
|
487
|
+
* as `stopped` rather than `done`, so an interrupted pass and a completed
|
|
488
|
+
* one do not read alike.
|
|
489
|
+
*
|
|
490
|
+
* A run that finished on its own is not rewritten: stopping one answers 409.
|
|
491
|
+
*
|
|
492
|
+
* Returns the stopped run itself, like `startRun` and `getRun`, not the
|
|
493
|
+
* `{run}` envelope the service sends. Every other single-run method in this
|
|
494
|
+
* class unwraps, and `Loop.stop_run` in the Python SDK does too, so a
|
|
495
|
+
* caller who wrote `(await loop.stopRun(id)).status` against one of its
|
|
496
|
+
* siblings is right here as well.
|
|
497
|
+
*/
|
|
498
|
+
stopRun(runId: string): Promise<LoopJudgeRun>;
|
|
430
499
|
/**
|
|
431
500
|
* Can an agent run here, is one running, and is it on for this workspace.
|
|
432
501
|
* `available` is about the environment; `online` about the worker;
|
|
@@ -494,6 +563,14 @@ export declare class Loop {
|
|
|
494
563
|
*
|
|
495
564
|
* Send the `terms_version` it returns back in `createTrainingRule`. Read the
|
|
496
565
|
* figures out of this response rather than inventing ceilings of your own.
|
|
566
|
+
*
|
|
567
|
+
* `valid: false` with an EMPTY `refusals` list is not nothing: check
|
|
568
|
+
* `unreachable`, which names every peer the platform could not reach. The
|
|
569
|
+
* rule is savable in that state and would be paused, but the estimate around
|
|
570
|
+
* it is not trustworthy -- both `worst_hourly_*` are `0` -- so do not quote
|
|
571
|
+
* those figures to anyone. `model_revision` is NOT a signal here: a degraded
|
|
572
|
+
* pass echoes back whatever revision the request carried, so test
|
|
573
|
+
* `unreachable.length > 0` rather than the revision being empty.
|
|
497
574
|
*/
|
|
498
575
|
preflightTrainingRule(params: TrainingRulePreflightRequest): Promise<TrainingRulePreflight>;
|
|
499
576
|
/**
|
package/dist/resources/loop.js
CHANGED
|
@@ -87,6 +87,37 @@ export class Loop {
|
|
|
87
87
|
* filed under: "everything in one bucket called import" is a corpus nobody
|
|
88
88
|
* can slice afterwards. At most 5000 rows per call, because the call is
|
|
89
89
|
* synchronous and somebody is waiting on it.
|
|
90
|
+
*
|
|
91
|
+
* WHAT COMES BACK. `imported` is what this call created and `trace_ids` names
|
|
92
|
+
* the conversations from the file that are now in the loop, so a caller can
|
|
93
|
+
* label, review or delete them. `by_shape`, `reviewed` and `needs_review`
|
|
94
|
+
* count what THIS CALL created, not what was merely recognised and not rows
|
|
95
|
+
* an earlier import already had, and the sentences in `notes` are written
|
|
96
|
+
* from those counts, so the numbers and the prose describe the same call and
|
|
97
|
+
* a re-send cannot report a reviewed conversation as waiting for review.
|
|
98
|
+
* `refused` is rows whose shape could not be read, which the file fixes;
|
|
99
|
+
* `not_saved` is rows that were understood and then could not be stored,
|
|
100
|
+
* which the file cannot fix.
|
|
101
|
+
*
|
|
102
|
+
* A call with any `not_saved`, or any `verdicts_not_saved` (the conversation
|
|
103
|
+
* stored and the judgement that came with it did not), REJECTS with
|
|
104
|
+
* `LoopImportIncompleteError`, whose `outcome` carries all of the above and
|
|
105
|
+
* whose `notSavedRows` and `verdictsNotSavedRows` name the positions.
|
|
106
|
+
*
|
|
107
|
+
* HOW A RETRY IS SAFE. Every response carries `import_id`, the token this
|
|
108
|
+
* call was filed under. Send the same rows again with it, and rows that
|
|
109
|
+
* already arrived come back on `already_present` instead of being stored
|
|
110
|
+
* twice, while a verdict that failed to write is attempted again. A call that
|
|
111
|
+
* repeats no token is its own import, deliberately: two calls carrying the
|
|
112
|
+
* same rows are as likely to be two pages of one export as one call twice,
|
|
113
|
+
* and guessing "retry" silently drops the second copy of every conversation a
|
|
114
|
+
* file lists more than once.
|
|
115
|
+
*
|
|
116
|
+
* SENDING A FILE IN PAGES. Either give each row its own `row_ids` entry, and
|
|
117
|
+
* then chunking, ordering and subsets all stop mattering; or send one
|
|
118
|
+
* `import_id` with the `row_offset` each page starts at. Do not send back
|
|
119
|
+
* only the rows `notSavedRows` names unless you used `row_ids`: without them
|
|
120
|
+
* a shorter list moves every row after the gap and imports it again.
|
|
90
121
|
*/
|
|
91
122
|
async importRows(params) {
|
|
92
123
|
return this._http.fetchPost('/api/loop/import', params);
|
|
@@ -280,7 +311,22 @@ export class Loop {
|
|
|
280
311
|
const q = key ? `?key=${encodeURIComponent(key)}` : '';
|
|
281
312
|
return this._http.fetchDelete(`/api/loop/traces/${encodeURIComponent(traceId)}/labels/${encodeURIComponent(label)}${q}`);
|
|
282
313
|
}
|
|
283
|
-
/**
|
|
314
|
+
/**
|
|
315
|
+
* Every label in the workspace, with how many conversations carry it and
|
|
316
|
+
* which master groups those conversations are in.
|
|
317
|
+
*
|
|
318
|
+
* TWO FIELDS ANSWER THE GROUP QUESTION, AND THEY ANSWER DIFFERENT ONES.
|
|
319
|
+
* `parent` is the group ALL of a label's conversations are in, and it is
|
|
320
|
+
* null the moment they disagree. `parents` is every group ANY of them are
|
|
321
|
+
* in, with how many of them are in each.
|
|
322
|
+
*
|
|
323
|
+
* Build a list of master groups out of `parents`. A label with 12
|
|
324
|
+
* conversations, 5 of them under `support`, reports `parent: null` and
|
|
325
|
+
* `parents: [{parent: 'support', traces: 5}]`, so code that reads only
|
|
326
|
+
* `parent` sees no group at all for it. The registry used to answer
|
|
327
|
+
* `parent: "support"` for all twelve, which named the group but put seven
|
|
328
|
+
* conversations in it that nobody had put there.
|
|
329
|
+
*/
|
|
284
330
|
async listLabels() {
|
|
285
331
|
const res = await this._http.fetchGet('/api/loop/labels');
|
|
286
332
|
return res.labels;
|
|
@@ -589,13 +635,39 @@ export class Loop {
|
|
|
589
635
|
* scoring something the rubric never asked for, is refused and reported in
|
|
590
636
|
* `rejected` rather than silently dropped.
|
|
591
637
|
*
|
|
592
|
-
* Pass `finish` to close
|
|
593
|
-
*
|
|
594
|
-
*
|
|
638
|
+
* Pass `finish` to close the run in the same call once you have nothing left
|
|
639
|
+
* to send. To close a run WITHOUT scores, call `stopRun`: an empty verdict
|
|
640
|
+
* list with `finish` does the same thing and reads like a mistake.
|
|
595
641
|
*/
|
|
596
642
|
async postVerdicts(runId, verdicts, finish = false) {
|
|
597
643
|
return this._http.fetchPost(`/api/loop/runs/${encodeURIComponent(runId)}/verdicts`, { verdicts, finish });
|
|
598
644
|
}
|
|
645
|
+
/**
|
|
646
|
+
* Close a run that has not finished.
|
|
647
|
+
*
|
|
648
|
+
* A judge may have one open run of its own at a time, so an open run BLOCKS
|
|
649
|
+
* the next one, and the runs that most need closing are the ones nobody can
|
|
650
|
+
* wait out: a run the agent has parked on an empty balance or a refused key
|
|
651
|
+
* stays open until the reason is fixed or somebody stops it.
|
|
652
|
+
*
|
|
653
|
+
* Nothing is deleted. Every verdict already recorded stays recorded, the
|
|
654
|
+
* counters keep saying how much of the selection was covered, and the
|
|
655
|
+
* conversations the run was holding are free for the next one. The run ends
|
|
656
|
+
* as `stopped` rather than `done`, so an interrupted pass and a completed
|
|
657
|
+
* one do not read alike.
|
|
658
|
+
*
|
|
659
|
+
* A run that finished on its own is not rewritten: stopping one answers 409.
|
|
660
|
+
*
|
|
661
|
+
* Returns the stopped run itself, like `startRun` and `getRun`, not the
|
|
662
|
+
* `{run}` envelope the service sends. Every other single-run method in this
|
|
663
|
+
* class unwraps, and `Loop.stop_run` in the Python SDK does too, so a
|
|
664
|
+
* caller who wrote `(await loop.stopRun(id)).status` against one of its
|
|
665
|
+
* siblings is right here as well.
|
|
666
|
+
*/
|
|
667
|
+
async stopRun(runId) {
|
|
668
|
+
const res = await this._http.fetchPost(`/api/loop/runs/${encodeURIComponent(runId)}/stop`, {});
|
|
669
|
+
return res.run;
|
|
670
|
+
}
|
|
599
671
|
// ── the agent ─────────────────────────────────────────────────────────
|
|
600
672
|
//
|
|
601
673
|
// The worker that calls a model on the workspace's behalf: it applies
|
|
@@ -700,6 +772,14 @@ export class Loop {
|
|
|
700
772
|
*
|
|
701
773
|
* Send the `terms_version` it returns back in `createTrainingRule`. Read the
|
|
702
774
|
* figures out of this response rather than inventing ceilings of your own.
|
|
775
|
+
*
|
|
776
|
+
* `valid: false` with an EMPTY `refusals` list is not nothing: check
|
|
777
|
+
* `unreachable`, which names every peer the platform could not reach. The
|
|
778
|
+
* rule is savable in that state and would be paused, but the estimate around
|
|
779
|
+
* it is not trustworthy -- both `worst_hourly_*` are `0` -- so do not quote
|
|
780
|
+
* those figures to anyone. `model_revision` is NOT a signal here: a degraded
|
|
781
|
+
* pass echoes back whatever revision the request carried, so test
|
|
782
|
+
* `unreachable.length > 0` rather than the revision being empty.
|
|
703
783
|
*/
|
|
704
784
|
async preflightTrainingRule(params) {
|
|
705
785
|
return this._http.fetchPost('/api/loop/training-rules/preflight', params);
|
package/dist/types.d.ts
CHANGED
|
@@ -2233,13 +2233,91 @@ export interface LoopImportParams {
|
|
|
2233
2233
|
/** Applied to every row: importing a dump is when somebody knows what it is. */
|
|
2234
2234
|
labels?: string[];
|
|
2235
2235
|
attributes?: Record<string, string>;
|
|
2236
|
+
/**
|
|
2237
|
+
* Repeat the `import_id` a previous call reported to make this call a RETRY
|
|
2238
|
+
* of that one: rows that already arrived come back under `already_present`
|
|
2239
|
+
* instead of being stored again.
|
|
2240
|
+
*
|
|
2241
|
+
* Leave it out and this call is its own import. The endpoint will NOT guess:
|
|
2242
|
+
* two calls carrying the same rows are as likely to be two pages of one
|
|
2243
|
+
* export as one call sent twice, and guessing "retry" silently drops the
|
|
2244
|
+
* second copy of every conversation a file lists more than once. Every
|
|
2245
|
+
* response carries the token it was filed under, so a retry is always
|
|
2246
|
+
* available and never has to be inferred.
|
|
2247
|
+
*
|
|
2248
|
+
* SEND THE ROWS AS YOU SENT THEM. Without `row_ids` a row is matched on its
|
|
2249
|
+
* text, and the match survives re-ordered keys and different whitespace but
|
|
2250
|
+
* NOT a number that has been re-spelled: `100`, `1e2` and `100.0` are three
|
|
2251
|
+
* different rows. A round trip through most JSON libraries re-spells numbers
|
|
2252
|
+
* (Python turns `1e2` into `100.0`), so a retry built by re-serialising a
|
|
2253
|
+
* parsed file can store rows a second time. Keep the bytes you sent, or send
|
|
2254
|
+
* `row_ids` and stop depending on the text at all.
|
|
2255
|
+
*/
|
|
2256
|
+
import_id?: string;
|
|
2257
|
+
/**
|
|
2258
|
+
* Where this page starts in the file, counting from 0. Send it when you split
|
|
2259
|
+
* one import across several calls under one `import_id`, so two pages are not
|
|
2260
|
+
* matched against each other row for row. Ignored when you send `row_ids`.
|
|
2261
|
+
*/
|
|
2262
|
+
row_offset?: number;
|
|
2263
|
+
/**
|
|
2264
|
+
* Your own id for each row, in the same order as `rows`: a ticket number, a
|
|
2265
|
+
* conversation id, whatever the export already carries.
|
|
2266
|
+
*
|
|
2267
|
+
* The strongest form of identity, and the one to reach for with a file big
|
|
2268
|
+
* enough to page. It SUPERSEDES `import_id` and `row_offset`: with it,
|
|
2269
|
+
* chunking, ordering and subsets all stop mattering and any part of the file
|
|
2270
|
+
* can be re-sent exactly. One per row or none at all, and no two rows in a
|
|
2271
|
+
* call may share an id: both are refused rather than silently merging two
|
|
2272
|
+
* conversations into one.
|
|
2273
|
+
*
|
|
2274
|
+
* THE ID IS THE IDENTITY, AND THE ROW'S TEXT IS NOT PART OF IT. A row sent
|
|
2275
|
+
* again under an id this source already imported comes back under
|
|
2276
|
+
* `already_present` and the stored conversation is left as it was, even if
|
|
2277
|
+
* you changed its text. A CORRECTION DOES NOT LAND THIS WAY: send it under a
|
|
2278
|
+
* new id. That is the trade for making every retry, page and subset safe -
|
|
2279
|
+
* the text is never compared, so nothing your serialiser does to it can turn
|
|
2280
|
+
* a retry into a second import.
|
|
2281
|
+
*/
|
|
2282
|
+
row_ids?: string[];
|
|
2236
2283
|
}
|
|
2284
|
+
/**
|
|
2285
|
+
* What an import did.
|
|
2286
|
+
*
|
|
2287
|
+
* A call where every row stored resolves with this. A call where SOME rows
|
|
2288
|
+
* could not be stored rejects with {@link LoopImportIncompleteError}, whose
|
|
2289
|
+
* `outcome` is this same shape: the counts are the whole answer either way, so
|
|
2290
|
+
* read them from the error exactly as you would from a success.
|
|
2291
|
+
*/
|
|
2237
2292
|
export interface LoopImportResult {
|
|
2293
|
+
/**
|
|
2294
|
+
* The token this call was filed under, whether you sent one or it was minted
|
|
2295
|
+
* for you. Send the same rows again with this `import_id` to retry the call.
|
|
2296
|
+
* Present on every response, success or failure, because a call can succeed
|
|
2297
|
+
* and still need retrying when the answer never reached you.
|
|
2298
|
+
*/
|
|
2299
|
+
import_id?: string;
|
|
2300
|
+
/** How many conversations this call created. */
|
|
2238
2301
|
imported: number;
|
|
2239
2302
|
/**
|
|
2240
|
-
* How many
|
|
2241
|
-
*
|
|
2242
|
-
*
|
|
2303
|
+
* How many rows an earlier call under the SAME `import_id` (or carrying the
|
|
2304
|
+
* same `row_ids`) had already imported, and so were not stored a second time.
|
|
2305
|
+
* Zero on a first import, and zero on any call that repeated neither: a call
|
|
2306
|
+
* that does not say it is a retry is its own import, which is what lets a
|
|
2307
|
+
* file sent in pages land complete.
|
|
2308
|
+
*/
|
|
2309
|
+
already_present?: number;
|
|
2310
|
+
/**
|
|
2311
|
+
* How many verdicts THIS CALL WROTE. Reported apart from `imported` because
|
|
2312
|
+
* it is the difference between data you can train on and data somebody still
|
|
2313
|
+
* has to look at.
|
|
2314
|
+
*
|
|
2315
|
+
* Re-sending a row whose verdict already landed writes nothing and counts
|
|
2316
|
+
* nothing, so a plain retry reads 0. Re-sending one that came back in
|
|
2317
|
+
* `verdicts_not_saved` DOES count here when the write succeeds this time,
|
|
2318
|
+
* even though the row itself is counted under `already_present`: that is the
|
|
2319
|
+
* first time anybody recorded that verdict, not a second reviewer. This is
|
|
2320
|
+
* the field to read to confirm a repair landed.
|
|
2243
2321
|
*/
|
|
2244
2322
|
reviewed: number;
|
|
2245
2323
|
needs_review: number;
|
|
@@ -2251,17 +2329,47 @@ export interface LoopImportResult {
|
|
|
2251
2329
|
/**
|
|
2252
2330
|
* How many rows were understood and then could not be stored. Counted apart
|
|
2253
2331
|
* from `refused`, because the two are different claims and rewriting a file
|
|
2254
|
-
* that was already correct fixes nothing. A call with any of these
|
|
2255
|
-
* rather than
|
|
2332
|
+
* that was already correct fixes nothing. A call with any of these rejects
|
|
2333
|
+
* rather than resolving; read it off {@link LoopImportIncompleteError.outcome}.
|
|
2256
2334
|
*/
|
|
2257
2335
|
not_saved?: number;
|
|
2258
|
-
/**
|
|
2336
|
+
/**
|
|
2337
|
+
* WHICH rows those were, counting from 1 in the order you sent them, so you
|
|
2338
|
+
* can say what is missing rather than only how much.
|
|
2339
|
+
*
|
|
2340
|
+
* NOT A SMALLER FILE TO SEND BACK unless you sent `row_ids`. Without them a
|
|
2341
|
+
* row is identified by its place in the call, so a shorter list moves every
|
|
2342
|
+
* row after the gap and each one is stored again. Re-send the whole set of
|
|
2343
|
+
* rows, in the same order, with the same `import_id`; what already arrived
|
|
2344
|
+
* comes back under `already_present`. With `row_ids` a row carries its own
|
|
2345
|
+
* identity and any subset is exact.
|
|
2346
|
+
*/
|
|
2347
|
+
not_saved_rows?: number[];
|
|
2348
|
+
/**
|
|
2349
|
+
* Rows whose CONVERSATION stored and whose VERDICT did not, and which rows
|
|
2350
|
+
* those were. A preference pair or a thumbs label is two writes, and the
|
|
2351
|
+
* second can fail on its own: the conversation is then in the loop carrying
|
|
2352
|
+
* nobody's judgement, counted under `needs_review` rather than `reviewed`,
|
|
2353
|
+
* and not trainable. The call rejects with {@link LoopImportIncompleteError},
|
|
2354
|
+
* and sending the rows again under the same `import_id` records the verdict
|
|
2355
|
+
* without storing the conversation twice.
|
|
2356
|
+
*/
|
|
2357
|
+
verdicts_not_saved?: number;
|
|
2358
|
+
verdicts_not_saved_rows?: number[];
|
|
2359
|
+
/**
|
|
2360
|
+
* How many conversations of each shape THIS CALL created. Rows that were
|
|
2361
|
+
* refused, rows that failed to store and rows an earlier import already had
|
|
2362
|
+
* are not in here: the counts and the sentences in `notes` describe the same
|
|
2363
|
+
* call, so neither can claim a conversation is waiting for review when none
|
|
2364
|
+
* was written.
|
|
2365
|
+
*/
|
|
2259
2366
|
by_shape: Record<string, number>;
|
|
2260
2367
|
refused_why: Record<string, number>;
|
|
2261
2368
|
/**
|
|
2262
|
-
* The conversations this
|
|
2263
|
-
*
|
|
2264
|
-
*
|
|
2369
|
+
* The conversations from this file that are now in the loop, in file order:
|
|
2370
|
+
* the ones this call created and the ones it found already there. Without
|
|
2371
|
+
* them a caller whose import half-failed has a number and no way to reach
|
|
2372
|
+
* what landed: no way to label it, review it, or delete it and start again.
|
|
2265
2373
|
*/
|
|
2266
2374
|
trace_ids?: string[];
|
|
2267
2375
|
notes: string[];
|
|
@@ -2350,7 +2458,11 @@ export interface LoopDatasetCreateParams {
|
|
|
2350
2458
|
to?: string;
|
|
2351
2459
|
/** Narrow to one slice. A label with children selects them too. */
|
|
2352
2460
|
label?: string;
|
|
2353
|
-
/**
|
|
2461
|
+
/**
|
|
2462
|
+
* Take this many of what the filters matched. Which ones is arbitrary but
|
|
2463
|
+
* REPEATABLE: the same conversations always give the same slice, so two
|
|
2464
|
+
* builds of one spec describe the same set.
|
|
2465
|
+
*/
|
|
2354
2466
|
sample?: number;
|
|
2355
2467
|
/** Holds back your most recent work, not a random slice. 0-50. */
|
|
2356
2468
|
holdout_percent?: number;
|
|
@@ -2535,8 +2647,13 @@ export interface LoopJudgeRun {
|
|
|
2535
2647
|
* it for a week and items were still waiting. Nothing is deleted and posting
|
|
2536
2648
|
* verdicts to it still works and still closes it as done. It exists so that
|
|
2537
2649
|
* `open` keeps meaning "somebody is working on this".
|
|
2650
|
+
*
|
|
2651
|
+
* `stopped` is a run somebody closed on purpose before it finished, with
|
|
2652
|
+
* `stopRun`. Separate from `done` because a run that covered three of forty
|
|
2653
|
+
* conversations did not finish its work: every score it recorded is kept
|
|
2654
|
+
* either way, and reading one as the other overstates what was evaluated.
|
|
2538
2655
|
*/
|
|
2539
|
-
status: 'open' | 'done' | 'abandoned';
|
|
2656
|
+
status: 'open' | 'done' | 'abandoned' | 'stopped';
|
|
2540
2657
|
instructions: string;
|
|
2541
2658
|
dimensions: LoopJudgeDimension[];
|
|
2542
2659
|
model: string | null;
|
|
@@ -2817,9 +2934,34 @@ export interface LoopLabel {
|
|
|
2817
2934
|
/** The dimension this value belongs to. `tag` for a bare name. */
|
|
2818
2935
|
key: string;
|
|
2819
2936
|
}
|
|
2937
|
+
/** One master group, and how many of a label's conversations are in it. */
|
|
2938
|
+
export interface LoopLabelParentCount {
|
|
2939
|
+
parent: string;
|
|
2940
|
+
traces: number;
|
|
2941
|
+
}
|
|
2820
2942
|
export interface LoopLabelCount {
|
|
2821
2943
|
label: string;
|
|
2944
|
+
/**
|
|
2945
|
+
* The master group ALL of this label's conversations are in, and nothing
|
|
2946
|
+
* else. Null when they disagree: the registry used to answer the
|
|
2947
|
+
* alphabetically last parent over a mixed group, so `billing` was reported
|
|
2948
|
+
* under `support` while seven of its twelve conversations were in no group
|
|
2949
|
+
* at all. Render "(in x)" from this field only.
|
|
2950
|
+
*/
|
|
2822
2951
|
parent: string | null;
|
|
2952
|
+
/**
|
|
2953
|
+
* Every master group ANY of them are in, sorted, with how many of this
|
|
2954
|
+
* label's conversations are in each. Empty when there are none.
|
|
2955
|
+
*
|
|
2956
|
+
* Read this, not `parent`, to discover which master groups exist: a label
|
|
2957
|
+
* whose conversations disagree still belongs partly to a real group, and
|
|
2958
|
+
* `parent` is null for it, so a picker built from `parent` alone loses the
|
|
2959
|
+
* group along with the false claim. The count is per group rather than the
|
|
2960
|
+
* label's own total, because adding a label's whole count to its parent is
|
|
2961
|
+
* the same mistake one level down. What is in no group at all is `traces`
|
|
2962
|
+
* minus the sum of these.
|
|
2963
|
+
*/
|
|
2964
|
+
parents: LoopLabelParentCount[];
|
|
2823
2965
|
traces: number;
|
|
2824
2966
|
key: string;
|
|
2825
2967
|
}
|
|
@@ -3002,6 +3144,22 @@ export interface TrainingRulePreflight {
|
|
|
3002
3144
|
eval_ceiling_cents: number;
|
|
3003
3145
|
warnings: string[];
|
|
3004
3146
|
refusals: TrainingRulePreflightRefusal[];
|
|
3147
|
+
/**
|
|
3148
|
+
* Every peer the platform could not reach on this pass, one entry each, in
|
|
3149
|
+
* the same `{stage, code, message}` shape as a refusal.
|
|
3150
|
+
*
|
|
3151
|
+
* A warning, not a refusal: an unreachable peer judged nothing, so it never
|
|
3152
|
+
* refuses a save. It is still why the rule cannot fire — both
|
|
3153
|
+
* `worst_hourly_*` come back `0` — so `valid` is `false`. Key a "cannot
|
|
3154
|
+
* save" state on this when `refusals` is empty; the same sentences are also
|
|
3155
|
+
* in `warnings`, so render one or the other.
|
|
3156
|
+
*
|
|
3157
|
+
* Key it on THIS, not on `model_revision`. A degraded pass echoes back the
|
|
3158
|
+
* `model_revision` the request carried rather than emptying it, so
|
|
3159
|
+
* `if (!model_revision)` is false on exactly the passes it was meant to
|
|
3160
|
+
* catch. Use `unreachable.length > 0`, or `/^[0-9a-f]{40}$/`.
|
|
3161
|
+
*/
|
|
3162
|
+
unreachable: TrainingRulePreflightRefusal[];
|
|
3005
3163
|
key_check: TrainingRuleKeyCheck;
|
|
3006
3164
|
terms_text: string;
|
|
3007
3165
|
terms_version: string;
|