runbios-sdk 0.2.1-dev.134 → 0.2.1-dev.136

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export declare const VERSION = "0.2.1-dev.134";
39
+ export declare const VERSION = "0.2.1-dev.136";
40
40
  export declare class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  readonly models: Models;
package/dist/index.js CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export const VERSION = '0.2.1-dev.134';
39
+ export const VERSION = '0.2.1-dev.136';
40
40
  export class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  models;
@@ -312,6 +312,13 @@ export declare class Loop {
312
312
  * twice the weight, and the combined score is the weighted mean over the
313
313
  * rules that actually applied.
314
314
  *
315
+ * `matches_gold` needs no `expected`: it compares each answer to the gold
316
+ * answer recorded on that same conversation, or the human correction when
317
+ * there is no gold, and scores sampled alternatives against the same gold.
318
+ * A conversation with neither is skipped, not failed, so one rule checks
319
+ * every labelled question without punishing the unlabelled ones. An
320
+ * optional `tolerance` widens the match when both sides are bare numbers.
321
+ *
315
322
  * A caution worth knowing before you write a set: a rule made only of
316
323
  * `forbidden` phrases is satisfied VACUOUSLY by an answer that says nothing.
317
324
  * Pair it with a `required` phrase, or you are rewarding silence.
@@ -408,8 +415,9 @@ export declare class Loop {
408
415
  */
409
416
  agentStatus(): Promise<LoopAgentStatus>;
410
417
  /**
411
- * Turn the agent on: mints the workspace's managed serverless key. After
412
- * this, automatic judges and sample runs make model calls billed to the
418
+ * Turn the agent on: mints the workspace's managed serverless key and
419
+ * resumes the automatic judges a previous turn-off paused. After this,
420
+ * automatic judges and sample runs make model calls billed to the
413
421
  * workspace. Idempotent.
414
422
  *
415
423
  * `monthly_spend_cap_cents` caps what that key may spend on model calls in
@@ -423,9 +431,11 @@ export declare class Loop {
423
431
  monthly_spend_cap_cents?: number;
424
432
  }): Promise<LoopAgentCredential>;
425
433
  /**
426
- * Turn the agent off: revokes its key and sets every automatic judge back to
427
- * manual. Open runs stop where they are and continue if it is turned back
428
- * on. Nothing already scored or written is removed.
434
+ * Turn the agent off: revokes its key and pauses every automatic judge in
435
+ * the workspace (`judges_paused`; each judge shows `auto_paused`). Turning
436
+ * it back on resumes exactly those judges. Open runs stop where they are
437
+ * and continue if it is turned back on. Nothing already scored or written
438
+ * is removed.
429
439
  */
430
440
  disableAgent(): Promise<{
431
441
  revoked: boolean;
@@ -417,6 +417,13 @@ export class Loop {
417
417
  * twice the weight, and the combined score is the weighted mean over the
418
418
  * rules that actually applied.
419
419
  *
420
+ * `matches_gold` needs no `expected`: it compares each answer to the gold
421
+ * answer recorded on that same conversation, or the human correction when
422
+ * there is no gold, and scores sampled alternatives against the same gold.
423
+ * A conversation with neither is skipped, not failed, so one rule checks
424
+ * every labelled question without punishing the unlabelled ones. An
425
+ * optional `tolerance` widens the match when both sides are bare numbers.
426
+ *
420
427
  * A caution worth knowing before you write a set: a rule made only of
421
428
  * `forbidden` phrases is satisfied VACUOUSLY by an answer that says nothing.
422
429
  * Pair it with a `required` phrase, or you are rewarding silence.
@@ -579,8 +586,9 @@ export class Loop {
579
586
  return this._http.fetchGet('/api/loop/agent');
580
587
  }
581
588
  /**
582
- * Turn the agent on: mints the workspace's managed serverless key. After
583
- * this, automatic judges and sample runs make model calls billed to the
589
+ * Turn the agent on: mints the workspace's managed serverless key and
590
+ * resumes the automatic judges a previous turn-off paused. After this,
591
+ * automatic judges and sample runs make model calls billed to the
584
592
  * workspace. Idempotent.
585
593
  *
586
594
  * `monthly_spend_cap_cents` caps what that key may spend on model calls in
@@ -598,9 +606,11 @@ export class Loop {
598
606
  return res.credential;
599
607
  }
600
608
  /**
601
- * Turn the agent off: revokes its key and sets every automatic judge back to
602
- * manual. Open runs stop where they are and continue if it is turned back
603
- * on. Nothing already scored or written is removed.
609
+ * Turn the agent off: revokes its key and pauses every automatic judge in
610
+ * the workspace (`judges_paused`; each judge shows `auto_paused`). Turning
611
+ * it back on resumes exactly those judges. Open runs stop where they are
612
+ * and continue if it is turned back on. Nothing already scored or written
613
+ * is removed.
604
614
  */
605
615
  async disableAgent() {
606
616
  return this._http.fetchDelete('/api/loop/agent');
package/dist/types.d.ts CHANGED
@@ -2672,8 +2672,16 @@ export interface LoopCandidate {
2672
2672
  metadata: Record<string, unknown>;
2673
2673
  created_at: string;
2674
2674
  }
2675
- /** The deterministic checks a grader can perform. */
2676
- export type LoopGraderKind = 'exact_match' | 'contains' | 'regex' | 'json_valid' | 'numeric' | 'tool_called';
2675
+ /**
2676
+ * The deterministic checks a grader can perform.
2677
+ *
2678
+ * `matches_gold` is the one kind with no expected value of its own: it
2679
+ * compares each answer to the gold answer recorded on that same conversation
2680
+ * (or the human correction when there is no gold), scores sampled
2681
+ * alternatives against the same gold, and is NOT APPLIED to a conversation
2682
+ * that carries neither, so it never marks down an unlabelled answer.
2683
+ */
2684
+ export type LoopGraderKind = 'matches_gold' | 'exact_match' | 'contains' | 'regex' | 'json_valid' | 'numeric' | 'tool_called';
2677
2685
  export interface LoopGraderConfig {
2678
2686
  expected?: string;
2679
2687
  /** Every phrase that must appear. */
@@ -2683,7 +2691,10 @@ export interface LoopGraderConfig {
2683
2691
  pattern?: string;
2684
2692
  /** Keys the answer must carry, for `json_valid`. */
2685
2693
  keys?: string[];
2686
- /** How far from `expected` still counts, for `numeric`. */
2694
+ /**
2695
+ * How far from `expected` still counts, for `numeric`. For `matches_gold`,
2696
+ * how far from the gold answer still counts when both are bare numbers.
2697
+ */
2687
2698
  tolerance?: number;
2688
2699
  /** The function that must have been called, for `tool_called`. */
2689
2700
  function?: string;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.1-dev.134",
3
+ "version": "0.2.1-dev.136",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",