runbios-sdk 0.2.1-rc.147 → 0.2.1-rc.153
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -1
- package/dist/index.js +1 -1
- package/dist/resources/loop.d.ts +51 -4
- package/dist/resources/loop.js +54 -4
- package/dist/types.d.ts +47 -1
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export declare const VERSION = "0.2.1-rc.
|
|
39
|
+
export declare const VERSION = "0.2.1-rc.153";
|
|
40
40
|
export declare class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
readonly models: Models;
|
package/dist/index.js
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export const VERSION = '0.2.1-rc.
|
|
39
|
+
export const VERSION = '0.2.1-rc.153';
|
|
40
40
|
export class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
models;
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -243,7 +243,22 @@ export declare class Loop {
|
|
|
243
243
|
key: string;
|
|
244
244
|
label: string;
|
|
245
245
|
}>;
|
|
246
|
-
/**
|
|
246
|
+
/**
|
|
247
|
+
* Every label in the workspace, with how many conversations carry it and
|
|
248
|
+
* which master groups those conversations are in.
|
|
249
|
+
*
|
|
250
|
+
* TWO FIELDS ANSWER THE GROUP QUESTION, AND THEY ANSWER DIFFERENT ONES.
|
|
251
|
+
* `parent` is the group ALL of a label's conversations are in, and it is
|
|
252
|
+
* null the moment they disagree. `parents` is every group ANY of them are
|
|
253
|
+
* in, with how many of them are in each.
|
|
254
|
+
*
|
|
255
|
+
* Build a list of master groups out of `parents`. A label with 12
|
|
256
|
+
* conversations, 5 of them under `support`, reports `parent: null` and
|
|
257
|
+
* `parents: [{parent: 'support', traces: 5}]`, so code that reads only
|
|
258
|
+
* `parent` sees no group at all for it. The registry used to answer
|
|
259
|
+
* `parent: "support"` for all twelve, which named the group but put seven
|
|
260
|
+
* conversations in it that nobody had put there.
|
|
261
|
+
*/
|
|
247
262
|
listLabels(): Promise<LoopLabelCount[]>;
|
|
248
263
|
/**
|
|
249
264
|
* Build a training set from the feedback recorded so far.
|
|
@@ -453,11 +468,34 @@ export declare class Loop {
|
|
|
453
468
|
* scoring something the rubric never asked for, is refused and reported in
|
|
454
469
|
* `rejected` rather than silently dropped.
|
|
455
470
|
*
|
|
456
|
-
* Pass `finish` to close
|
|
457
|
-
*
|
|
458
|
-
*
|
|
471
|
+
* Pass `finish` to close the run in the same call once you have nothing left
|
|
472
|
+
* to send. To close a run WITHOUT scores, call `stopRun`: an empty verdict
|
|
473
|
+
* list with `finish` does the same thing and reads like a mistake.
|
|
459
474
|
*/
|
|
460
475
|
postVerdicts(runId: string, verdicts: LoopJudgeVerdict[], finish?: boolean): Promise<LoopJudgeVerdictResult>;
|
|
476
|
+
/**
|
|
477
|
+
* Close a run that has not finished.
|
|
478
|
+
*
|
|
479
|
+
* A judge may have one open run of its own at a time, so an open run BLOCKS
|
|
480
|
+
* the next one, and the runs that most need closing are the ones nobody can
|
|
481
|
+
* wait out: a run the agent has parked on an empty balance or a refused key
|
|
482
|
+
* stays open until the reason is fixed or somebody stops it.
|
|
483
|
+
*
|
|
484
|
+
* Nothing is deleted. Every verdict already recorded stays recorded, the
|
|
485
|
+
* counters keep saying how much of the selection was covered, and the
|
|
486
|
+
* conversations the run was holding are free for the next one. The run ends
|
|
487
|
+
* as `stopped` rather than `done`, so an interrupted pass and a completed
|
|
488
|
+
* one do not read alike.
|
|
489
|
+
*
|
|
490
|
+
* A run that finished on its own is not rewritten: stopping one answers 409.
|
|
491
|
+
*
|
|
492
|
+
* Returns the stopped run itself, like `startRun` and `getRun`, not the
|
|
493
|
+
* `{run}` envelope the service sends. Every other single-run method in this
|
|
494
|
+
* class unwraps, and `Loop.stop_run` in the Python SDK does too, so a
|
|
495
|
+
* caller who wrote `(await loop.stopRun(id)).status` against one of its
|
|
496
|
+
* siblings is right here as well.
|
|
497
|
+
*/
|
|
498
|
+
stopRun(runId: string): Promise<LoopJudgeRun>;
|
|
461
499
|
/**
|
|
462
500
|
* Can an agent run here, is one running, and is it on for this workspace.
|
|
463
501
|
* `available` is about the environment; `online` about the worker;
|
|
@@ -525,6 +563,15 @@ export declare class Loop {
|
|
|
525
563
|
*
|
|
526
564
|
* Send the `terms_version` it returns back in `createTrainingRule`. Read the
|
|
527
565
|
* figures out of this response rather than inventing ceilings of your own.
|
|
566
|
+
*
|
|
567
|
+
* `valid: false` with an EMPTY `refusals` list is not nothing: check
|
|
568
|
+
* `unreachable`, which names every peer the platform could not reach. The
|
|
569
|
+
* rule is savable in that state and would be paused, but the estimate around
|
|
570
|
+
* it is not trustworthy -- both `worst_hourly_*` are `0`, and so are both
|
|
571
|
+
* `max_*_hours` beside them -- so do not quote those figures to anyone.
|
|
572
|
+
* `model_revision` is empty unless training-service actually pinned a
|
|
573
|
+
* commit; test `unreachable.length > 0` to tell a degraded pass from a
|
|
574
|
+
* refusal.
|
|
528
575
|
*/
|
|
529
576
|
preflightTrainingRule(params: TrainingRulePreflightRequest): Promise<TrainingRulePreflight>;
|
|
530
577
|
/**
|
package/dist/resources/loop.js
CHANGED
|
@@ -311,7 +311,22 @@ export class Loop {
|
|
|
311
311
|
const q = key ? `?key=${encodeURIComponent(key)}` : '';
|
|
312
312
|
return this._http.fetchDelete(`/api/loop/traces/${encodeURIComponent(traceId)}/labels/${encodeURIComponent(label)}${q}`);
|
|
313
313
|
}
|
|
314
|
-
/**
|
|
314
|
+
/**
|
|
315
|
+
* Every label in the workspace, with how many conversations carry it and
|
|
316
|
+
* which master groups those conversations are in.
|
|
317
|
+
*
|
|
318
|
+
* TWO FIELDS ANSWER THE GROUP QUESTION, AND THEY ANSWER DIFFERENT ONES.
|
|
319
|
+
* `parent` is the group ALL of a label's conversations are in, and it is
|
|
320
|
+
* null the moment they disagree. `parents` is every group ANY of them are
|
|
321
|
+
* in, with how many of them are in each.
|
|
322
|
+
*
|
|
323
|
+
* Build a list of master groups out of `parents`. A label with 12
|
|
324
|
+
* conversations, 5 of them under `support`, reports `parent: null` and
|
|
325
|
+
* `parents: [{parent: 'support', traces: 5}]`, so code that reads only
|
|
326
|
+
* `parent` sees no group at all for it. The registry used to answer
|
|
327
|
+
* `parent: "support"` for all twelve, which named the group but put seven
|
|
328
|
+
* conversations in it that nobody had put there.
|
|
329
|
+
*/
|
|
315
330
|
async listLabels() {
|
|
316
331
|
const res = await this._http.fetchGet('/api/loop/labels');
|
|
317
332
|
return res.labels;
|
|
@@ -620,13 +635,39 @@ export class Loop {
|
|
|
620
635
|
* scoring something the rubric never asked for, is refused and reported in
|
|
621
636
|
* `rejected` rather than silently dropped.
|
|
622
637
|
*
|
|
623
|
-
* Pass `finish` to close
|
|
624
|
-
*
|
|
625
|
-
*
|
|
638
|
+
* Pass `finish` to close the run in the same call once you have nothing left
|
|
639
|
+
* to send. To close a run WITHOUT scores, call `stopRun`: an empty verdict
|
|
640
|
+
* list with `finish` does the same thing and reads like a mistake.
|
|
626
641
|
*/
|
|
627
642
|
async postVerdicts(runId, verdicts, finish = false) {
|
|
628
643
|
return this._http.fetchPost(`/api/loop/runs/${encodeURIComponent(runId)}/verdicts`, { verdicts, finish });
|
|
629
644
|
}
|
|
645
|
+
/**
|
|
646
|
+
* Close a run that has not finished.
|
|
647
|
+
*
|
|
648
|
+
* A judge may have one open run of its own at a time, so an open run BLOCKS
|
|
649
|
+
* the next one, and the runs that most need closing are the ones nobody can
|
|
650
|
+
* wait out: a run the agent has parked on an empty balance or a refused key
|
|
651
|
+
* stays open until the reason is fixed or somebody stops it.
|
|
652
|
+
*
|
|
653
|
+
* Nothing is deleted. Every verdict already recorded stays recorded, the
|
|
654
|
+
* counters keep saying how much of the selection was covered, and the
|
|
655
|
+
* conversations the run was holding are free for the next one. The run ends
|
|
656
|
+
* as `stopped` rather than `done`, so an interrupted pass and a completed
|
|
657
|
+
* one do not read alike.
|
|
658
|
+
*
|
|
659
|
+
* A run that finished on its own is not rewritten: stopping one answers 409.
|
|
660
|
+
*
|
|
661
|
+
* Returns the stopped run itself, like `startRun` and `getRun`, not the
|
|
662
|
+
* `{run}` envelope the service sends. Every other single-run method in this
|
|
663
|
+
* class unwraps, and `Loop.stop_run` in the Python SDK does too, so a
|
|
664
|
+
* caller who wrote `(await loop.stopRun(id)).status` against one of its
|
|
665
|
+
* siblings is right here as well.
|
|
666
|
+
*/
|
|
667
|
+
async stopRun(runId) {
|
|
668
|
+
const res = await this._http.fetchPost(`/api/loop/runs/${encodeURIComponent(runId)}/stop`, {});
|
|
669
|
+
return res.run;
|
|
670
|
+
}
|
|
630
671
|
// ── the agent ─────────────────────────────────────────────────────────
|
|
631
672
|
//
|
|
632
673
|
// The worker that calls a model on the workspace's behalf: it applies
|
|
@@ -731,6 +772,15 @@ export class Loop {
|
|
|
731
772
|
*
|
|
732
773
|
* Send the `terms_version` it returns back in `createTrainingRule`. Read the
|
|
733
774
|
* figures out of this response rather than inventing ceilings of your own.
|
|
775
|
+
*
|
|
776
|
+
* `valid: false` with an EMPTY `refusals` list is not nothing: check
|
|
777
|
+
* `unreachable`, which names every peer the platform could not reach. The
|
|
778
|
+
* rule is savable in that state and would be paused, but the estimate around
|
|
779
|
+
* it is not trustworthy -- both `worst_hourly_*` are `0`, and so are both
|
|
780
|
+
* `max_*_hours` beside them -- so do not quote those figures to anyone.
|
|
781
|
+
* `model_revision` is empty unless training-service actually pinned a
|
|
782
|
+
* commit; test `unreachable.length > 0` to tell a degraded pass from a
|
|
783
|
+
* refusal.
|
|
734
784
|
*/
|
|
735
785
|
async preflightTrainingRule(params) {
|
|
736
786
|
return this._http.fetchPost('/api/loop/training-rules/preflight', params);
|
package/dist/types.d.ts
CHANGED
|
@@ -2647,8 +2647,13 @@ export interface LoopJudgeRun {
|
|
|
2647
2647
|
* it for a week and items were still waiting. Nothing is deleted and posting
|
|
2648
2648
|
* verdicts to it still works and still closes it as done. It exists so that
|
|
2649
2649
|
* `open` keeps meaning "somebody is working on this".
|
|
2650
|
+
*
|
|
2651
|
+
* `stopped` is a run somebody closed on purpose before it finished, with
|
|
2652
|
+
* `stopRun`. Separate from `done` because a run that covered three of forty
|
|
2653
|
+
* conversations did not finish its work: every score it recorded is kept
|
|
2654
|
+
* either way, and reading one as the other overstates what was evaluated.
|
|
2650
2655
|
*/
|
|
2651
|
-
status: 'open' | 'done' | 'abandoned';
|
|
2656
|
+
status: 'open' | 'done' | 'abandoned' | 'stopped';
|
|
2652
2657
|
instructions: string;
|
|
2653
2658
|
dimensions: LoopJudgeDimension[];
|
|
2654
2659
|
model: string | null;
|
|
@@ -2929,9 +2934,34 @@ export interface LoopLabel {
|
|
|
2929
2934
|
/** The dimension this value belongs to. `tag` for a bare name. */
|
|
2930
2935
|
key: string;
|
|
2931
2936
|
}
|
|
2937
|
+
/** One master group, and how many of a label's conversations are in it. */
|
|
2938
|
+
export interface LoopLabelParentCount {
|
|
2939
|
+
parent: string;
|
|
2940
|
+
traces: number;
|
|
2941
|
+
}
|
|
2932
2942
|
export interface LoopLabelCount {
|
|
2933
2943
|
label: string;
|
|
2944
|
+
/**
|
|
2945
|
+
* The master group ALL of this label's conversations are in, and nothing
|
|
2946
|
+
* else. Null when they disagree: the registry used to answer the
|
|
2947
|
+
* alphabetically last parent over a mixed group, so `billing` was reported
|
|
2948
|
+
* under `support` while seven of its twelve conversations were in no group
|
|
2949
|
+
* at all. Render "(in x)" from this field only.
|
|
2950
|
+
*/
|
|
2934
2951
|
parent: string | null;
|
|
2952
|
+
/**
|
|
2953
|
+
* Every master group ANY of them are in, sorted, with how many of this
|
|
2954
|
+
* label's conversations are in each. Empty when there are none.
|
|
2955
|
+
*
|
|
2956
|
+
* Read this, not `parent`, to discover which master groups exist: a label
|
|
2957
|
+
* whose conversations disagree still belongs partly to a real group, and
|
|
2958
|
+
* `parent` is null for it, so a picker built from `parent` alone loses the
|
|
2959
|
+
* group along with the false claim. The count is per group rather than the
|
|
2960
|
+
* label's own total, because adding a label's whole count to its parent is
|
|
2961
|
+
* the same mistake one level down. What is in no group at all is `traces`
|
|
2962
|
+
* minus the sum of these.
|
|
2963
|
+
*/
|
|
2964
|
+
parents: LoopLabelParentCount[];
|
|
2935
2965
|
traces: number;
|
|
2936
2966
|
key: string;
|
|
2937
2967
|
}
|
|
@@ -3114,6 +3144,22 @@ export interface TrainingRulePreflight {
|
|
|
3114
3144
|
eval_ceiling_cents: number;
|
|
3115
3145
|
warnings: string[];
|
|
3116
3146
|
refusals: TrainingRulePreflightRefusal[];
|
|
3147
|
+
/**
|
|
3148
|
+
* Every peer the platform could not reach on this pass, one entry each, in
|
|
3149
|
+
* the same `{stage, code, message}` shape as a refusal.
|
|
3150
|
+
*
|
|
3151
|
+
* A warning, not a refusal: an unreachable peer judged nothing, so it never
|
|
3152
|
+
* refuses a save. It is still why the rule cannot fire — both
|
|
3153
|
+
* `worst_hourly_*` come back `0` — so `valid` is `false`. Key a "cannot
|
|
3154
|
+
* save" state on this when `refusals` is empty; the same sentences are also
|
|
3155
|
+
* in `warnings`, so render one or the other.
|
|
3156
|
+
*
|
|
3157
|
+
* Key it on THIS, not on `model_revision`. A degraded pass echoes back the
|
|
3158
|
+
* `model_revision` the request carried rather than emptying it, so
|
|
3159
|
+
* `if (!model_revision)` is false on exactly the passes it was meant to
|
|
3160
|
+
* catch. Use `unreachable.length > 0`, or `/^[0-9a-f]{40}$/`.
|
|
3161
|
+
*/
|
|
3162
|
+
unreachable: TrainingRulePreflightRefusal[];
|
|
3117
3163
|
key_check: TrainingRuleKeyCheck;
|
|
3118
3164
|
terms_text: string;
|
|
3119
3165
|
terms_version: string;
|