runbios-sdk 0.2.1-dev.134 → 0.2.1-dev.135
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -1
- package/dist/index.js +1 -1
- package/dist/resources/loop.d.ts +7 -0
- package/dist/resources/loop.js +7 -0
- package/dist/types.d.ts +14 -3
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export declare const VERSION = "0.2.1-dev.
|
|
39
|
+
export declare const VERSION = "0.2.1-dev.135";
|
|
40
40
|
export declare class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
readonly models: Models;
|
package/dist/index.js
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export const VERSION = '0.2.1-dev.
|
|
39
|
+
export const VERSION = '0.2.1-dev.135';
|
|
40
40
|
export class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
models;
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -312,6 +312,13 @@ export declare class Loop {
|
|
|
312
312
|
* twice the weight, and the combined score is the weighted mean over the
|
|
313
313
|
* rules that actually applied.
|
|
314
314
|
*
|
|
315
|
+
* `matches_gold` needs no `expected`: it compares each answer to the gold
|
|
316
|
+
* answer recorded on that same conversation, or the human correction when
|
|
317
|
+
* there is no gold, and scores sampled alternatives against the same gold.
|
|
318
|
+
* A conversation with neither is skipped, not failed, so one rule checks
|
|
319
|
+
* every labelled question without punishing the unlabelled ones. An
|
|
320
|
+
* optional `tolerance` widens the match when both sides are bare numbers.
|
|
321
|
+
*
|
|
315
322
|
* A caution worth knowing before you write a set: a rule made only of
|
|
316
323
|
* `forbidden` phrases is satisfied VACUOUSLY by an answer that says nothing.
|
|
317
324
|
* Pair it with a `required` phrase, or you are rewarding silence.
|
package/dist/resources/loop.js
CHANGED
|
@@ -417,6 +417,13 @@ export class Loop {
|
|
|
417
417
|
* twice the weight, and the combined score is the weighted mean over the
|
|
418
418
|
* rules that actually applied.
|
|
419
419
|
*
|
|
420
|
+
* `matches_gold` needs no `expected`: it compares each answer to the gold
|
|
421
|
+
* answer recorded on that same conversation, or the human correction when
|
|
422
|
+
* there is no gold, and scores sampled alternatives against the same gold.
|
|
423
|
+
* A conversation with neither is skipped, not failed, so one rule checks
|
|
424
|
+
* every labelled question without punishing the unlabelled ones. An
|
|
425
|
+
* optional `tolerance` widens the match when both sides are bare numbers.
|
|
426
|
+
*
|
|
420
427
|
* A caution worth knowing before you write a set: a rule made only of
|
|
421
428
|
* `forbidden` phrases is satisfied VACUOUSLY by an answer that says nothing.
|
|
422
429
|
* Pair it with a `required` phrase, or you are rewarding silence.
|
package/dist/types.d.ts
CHANGED
|
@@ -2672,8 +2672,16 @@ export interface LoopCandidate {
|
|
|
2672
2672
|
metadata: Record<string, unknown>;
|
|
2673
2673
|
created_at: string;
|
|
2674
2674
|
}
|
|
2675
|
-
/**
|
|
2676
|
-
|
|
2675
|
+
/**
|
|
2676
|
+
* The deterministic checks a grader can perform.
|
|
2677
|
+
*
|
|
2678
|
+
* `matches_gold` is the one kind with no expected value of its own: it
|
|
2679
|
+
* compares each answer to the gold answer recorded on that same conversation
|
|
2680
|
+
* (or the human correction when there is no gold), scores sampled
|
|
2681
|
+
* alternatives against the same gold, and is NOT APPLIED to a conversation
|
|
2682
|
+
* that carries neither, so it never marks down an unlabelled answer.
|
|
2683
|
+
*/
|
|
2684
|
+
export type LoopGraderKind = 'matches_gold' | 'exact_match' | 'contains' | 'regex' | 'json_valid' | 'numeric' | 'tool_called';
|
|
2677
2685
|
export interface LoopGraderConfig {
|
|
2678
2686
|
expected?: string;
|
|
2679
2687
|
/** Every phrase that must appear. */
|
|
@@ -2683,7 +2691,10 @@ export interface LoopGraderConfig {
|
|
|
2683
2691
|
pattern?: string;
|
|
2684
2692
|
/** Keys the answer must carry, for `json_valid`. */
|
|
2685
2693
|
keys?: string[];
|
|
2686
|
-
/**
|
|
2694
|
+
/**
|
|
2695
|
+
* How far from `expected` still counts, for `numeric`. For `matches_gold`,
|
|
2696
|
+
* how far from the gold answer still counts when both are bare numbers.
|
|
2697
|
+
*/
|
|
2687
2698
|
tolerance?: number;
|
|
2688
2699
|
/** The function that must have been called, for `tool_called`. */
|
|
2689
2700
|
function?: string;
|