runbios-sdk 0.2.1-dev.134 → 0.2.1-dev.136
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -1
- package/dist/index.js +1 -1
- package/dist/resources/loop.d.ts +15 -5
- package/dist/resources/loop.js +15 -5
- package/dist/types.d.ts +14 -3
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export declare const VERSION = "0.2.1-dev.
|
|
39
|
+
export declare const VERSION = "0.2.1-dev.136";
|
|
40
40
|
export declare class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
readonly models: Models;
|
package/dist/index.js
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export const VERSION = '0.2.1-dev.
|
|
39
|
+
export const VERSION = '0.2.1-dev.136';
|
|
40
40
|
export class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
models;
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -312,6 +312,13 @@ export declare class Loop {
|
|
|
312
312
|
* twice the weight, and the combined score is the weighted mean over the
|
|
313
313
|
* rules that actually applied.
|
|
314
314
|
*
|
|
315
|
+
* `matches_gold` needs no `expected`: it compares each answer to the gold
|
|
316
|
+
* answer recorded on that same conversation, or the human correction when
|
|
317
|
+
* there is no gold, and scores sampled alternatives against the same gold.
|
|
318
|
+
* A conversation with neither is skipped, not failed, so one rule checks
|
|
319
|
+
* every labelled question without punishing the unlabelled ones. An
|
|
320
|
+
* optional `tolerance` widens the match when both sides are bare numbers.
|
|
321
|
+
*
|
|
315
322
|
* A caution worth knowing before you write a set: a rule made only of
|
|
316
323
|
* `forbidden` phrases is satisfied VACUOUSLY by an answer that says nothing.
|
|
317
324
|
* Pair it with a `required` phrase, or you are rewarding silence.
|
|
@@ -408,8 +415,9 @@ export declare class Loop {
|
|
|
408
415
|
*/
|
|
409
416
|
agentStatus(): Promise<LoopAgentStatus>;
|
|
410
417
|
/**
|
|
411
|
-
* Turn the agent on: mints the workspace's managed serverless key
|
|
412
|
-
*
|
|
418
|
+
* Turn the agent on: mints the workspace's managed serverless key and
|
|
419
|
+
* resumes the automatic judges a previous turn-off paused. After this,
|
|
420
|
+
* automatic judges and sample runs make model calls billed to the
|
|
413
421
|
* workspace. Idempotent.
|
|
414
422
|
*
|
|
415
423
|
* `monthly_spend_cap_cents` caps what that key may spend on model calls in
|
|
@@ -423,9 +431,11 @@ export declare class Loop {
|
|
|
423
431
|
monthly_spend_cap_cents?: number;
|
|
424
432
|
}): Promise<LoopAgentCredential>;
|
|
425
433
|
/**
|
|
426
|
-
* Turn the agent off: revokes its key and
|
|
427
|
-
*
|
|
428
|
-
* on.
|
|
434
|
+
* Turn the agent off: revokes its key and pauses every automatic judge in
|
|
435
|
+
* the workspace (`judges_paused`; each judge shows `auto_paused`). Turning
|
|
436
|
+
* it back on resumes exactly those judges. Open runs stop where they are
|
|
437
|
+
* and continue if it is turned back on. Nothing already scored or written
|
|
438
|
+
* is removed.
|
|
429
439
|
*/
|
|
430
440
|
disableAgent(): Promise<{
|
|
431
441
|
revoked: boolean;
|
package/dist/resources/loop.js
CHANGED
|
@@ -417,6 +417,13 @@ export class Loop {
|
|
|
417
417
|
* twice the weight, and the combined score is the weighted mean over the
|
|
418
418
|
* rules that actually applied.
|
|
419
419
|
*
|
|
420
|
+
* `matches_gold` needs no `expected`: it compares each answer to the gold
|
|
421
|
+
* answer recorded on that same conversation, or the human correction when
|
|
422
|
+
* there is no gold, and scores sampled alternatives against the same gold.
|
|
423
|
+
* A conversation with neither is skipped, not failed, so one rule checks
|
|
424
|
+
* every labelled question without punishing the unlabelled ones. An
|
|
425
|
+
* optional `tolerance` widens the match when both sides are bare numbers.
|
|
426
|
+
*
|
|
420
427
|
* A caution worth knowing before you write a set: a rule made only of
|
|
421
428
|
* `forbidden` phrases is satisfied VACUOUSLY by an answer that says nothing.
|
|
422
429
|
* Pair it with a `required` phrase, or you are rewarding silence.
|
|
@@ -579,8 +586,9 @@ export class Loop {
|
|
|
579
586
|
return this._http.fetchGet('/api/loop/agent');
|
|
580
587
|
}
|
|
581
588
|
/**
|
|
582
|
-
* Turn the agent on: mints the workspace's managed serverless key
|
|
583
|
-
*
|
|
589
|
+
* Turn the agent on: mints the workspace's managed serverless key and
|
|
590
|
+
* resumes the automatic judges a previous turn-off paused. After this,
|
|
591
|
+
* automatic judges and sample runs make model calls billed to the
|
|
584
592
|
* workspace. Idempotent.
|
|
585
593
|
*
|
|
586
594
|
* `monthly_spend_cap_cents` caps what that key may spend on model calls in
|
|
@@ -598,9 +606,11 @@ export class Loop {
|
|
|
598
606
|
return res.credential;
|
|
599
607
|
}
|
|
600
608
|
/**
|
|
601
|
-
* Turn the agent off: revokes its key and
|
|
602
|
-
*
|
|
603
|
-
* on.
|
|
609
|
+
* Turn the agent off: revokes its key and pauses every automatic judge in
|
|
610
|
+
* the workspace (`judges_paused`; each judge shows `auto_paused`). Turning
|
|
611
|
+
* it back on resumes exactly those judges. Open runs stop where they are
|
|
612
|
+
* and continue if it is turned back on. Nothing already scored or written
|
|
613
|
+
* is removed.
|
|
604
614
|
*/
|
|
605
615
|
async disableAgent() {
|
|
606
616
|
return this._http.fetchDelete('/api/loop/agent');
|
package/dist/types.d.ts
CHANGED
|
@@ -2672,8 +2672,16 @@ export interface LoopCandidate {
|
|
|
2672
2672
|
metadata: Record<string, unknown>;
|
|
2673
2673
|
created_at: string;
|
|
2674
2674
|
}
|
|
2675
|
-
/**
|
|
2676
|
-
|
|
2675
|
+
/**
|
|
2676
|
+
* The deterministic checks a grader can perform.
|
|
2677
|
+
*
|
|
2678
|
+
* `matches_gold` is the one kind with no expected value of its own: it
|
|
2679
|
+
* compares each answer to the gold answer recorded on that same conversation
|
|
2680
|
+
* (or the human correction when there is no gold), scores sampled
|
|
2681
|
+
* alternatives against the same gold, and is NOT APPLIED to a conversation
|
|
2682
|
+
* that carries neither, so it never marks down an unlabelled answer.
|
|
2683
|
+
*/
|
|
2684
|
+
export type LoopGraderKind = 'matches_gold' | 'exact_match' | 'contains' | 'regex' | 'json_valid' | 'numeric' | 'tool_called';
|
|
2677
2685
|
export interface LoopGraderConfig {
|
|
2678
2686
|
expected?: string;
|
|
2679
2687
|
/** Every phrase that must appear. */
|
|
@@ -2683,7 +2691,10 @@ export interface LoopGraderConfig {
|
|
|
2683
2691
|
pattern?: string;
|
|
2684
2692
|
/** Keys the answer must carry, for `json_valid`. */
|
|
2685
2693
|
keys?: string[];
|
|
2686
|
-
/**
|
|
2694
|
+
/**
|
|
2695
|
+
* How far from `expected` still counts, for `numeric`. For `matches_gold`,
|
|
2696
|
+
* how far from the gold answer still counts when both are bare numbers.
|
|
2697
|
+
*/
|
|
2687
2698
|
tolerance?: number;
|
|
2688
2699
|
/** The function that must have been called, for `tool_called`. */
|
|
2689
2700
|
function?: string;
|