runbios-sdk 0.2.1-dev.133 → 0.2.1-dev.135

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export declare const VERSION = "0.2.1-dev.133";
39
+ export declare const VERSION = "0.2.1-dev.135";
40
40
  export declare class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  readonly models: Models;
package/dist/index.js CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export const VERSION = '0.2.1-dev.133';
39
+ export const VERSION = '0.2.1-dev.135';
40
40
  export class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  models;
@@ -312,6 +312,13 @@ export declare class Loop {
312
312
  * twice the weight, and the combined score is the weighted mean over the
313
313
  * rules that actually applied.
314
314
  *
315
+ * `matches_gold` needs no `expected`: it compares each answer to the gold
316
+ * answer recorded on that same conversation, or the human correction when
317
+ * there is no gold, and scores sampled alternatives against the same gold.
318
+ * A conversation with neither is skipped, not failed, so one rule checks
319
+ * every labelled question without punishing the unlabelled ones. An
320
+ * optional `tolerance` widens the match when both sides are bare numbers.
321
+ *
315
322
  * A caution worth knowing before you write a set: a rule made only of
316
323
  * `forbidden` phrases is satisfied VACUOUSLY by an answer that says nothing.
317
324
  * Pair it with a `required` phrase, or you are rewarding silence.
@@ -411,8 +418,17 @@ export declare class Loop {
411
418
  * Turn the agent on: mints the workspace's managed serverless key. After
412
419
  * this, automatic judges and sample runs make model calls billed to the
413
420
  * workspace. Idempotent.
421
+ *
422
+ * `monthly_spend_cap_cents` caps what that key may spend on model calls in
423
+ * a calendar month. Omitted = no cap on a fresh key, and an existing cap is
424
+ * left as it is; sent while the agent is already on, it moves the cap on
425
+ * the existing key without minting a new one. When the cap is reached the
426
+ * agent's model calls are refused until next month and its runs pause with
427
+ * that reason.
414
428
  */
415
- enableAgent(): Promise<LoopAgentCredential>;
429
+ enableAgent(opts?: {
430
+ monthly_spend_cap_cents?: number;
431
+ }): Promise<LoopAgentCredential>;
416
432
  /**
417
433
  * Turn the agent off: revokes its key and sets every automatic judge back to
418
434
  * manual. Open runs stop where they are and continue if it is turned back
@@ -417,6 +417,13 @@ export class Loop {
417
417
  * twice the weight, and the combined score is the weighted mean over the
418
418
  * rules that actually applied.
419
419
  *
420
+ * `matches_gold` needs no `expected`: it compares each answer to the gold
421
+ * answer recorded on that same conversation, or the human correction when
422
+ * there is no gold, and scores sampled alternatives against the same gold.
423
+ * A conversation with neither is skipped, not failed, so one rule checks
424
+ * every labelled question without punishing the unlabelled ones. An
425
+ * optional `tolerance` widens the match when both sides are bare numbers.
426
+ *
420
427
  * A caution worth knowing before you write a set: a rule made only of
421
428
  * `forbidden` phrases is satisfied VACUOUSLY by an answer that says nothing.
422
429
  * Pair it with a `required` phrase, or you are rewarding silence.
@@ -582,9 +589,19 @@ export class Loop {
582
589
  * Turn the agent on: mints the workspace's managed serverless key. After
583
590
  * this, automatic judges and sample runs make model calls billed to the
584
591
  * workspace. Idempotent.
592
+ *
593
+ * `monthly_spend_cap_cents` caps what that key may spend on model calls in
594
+ * a calendar month. Omitted = no cap on a fresh key, and an existing cap is
595
+ * left as it is; sent while the agent is already on, it moves the cap on
596
+ * the existing key without minting a new one. When the cap is reached the
597
+ * agent's model calls are refused until next month and its runs pause with
598
+ * that reason.
585
599
  */
586
- async enableAgent() {
587
- const res = await this._http.fetchPost('/api/loop/agent', {});
600
+ async enableAgent(opts = {}) {
601
+ const body = {};
602
+ if (opts.monthly_spend_cap_cents != null)
603
+ body.monthly_spend_cap_cents = opts.monthly_spend_cap_cents;
604
+ const res = await this._http.fetchPost('/api/loop/agent', body);
588
605
  return res.credential;
589
606
  }
590
607
  /**
package/dist/types.d.ts CHANGED
@@ -2527,6 +2527,13 @@ export interface LoopAgentCredential {
2527
2527
  last_used_at: string | null;
2528
2528
  last_error: string | null;
2529
2529
  revoked_at: string | null;
2530
+ /**
2531
+ * The most the agent's key may spend on model calls in a calendar month,
2532
+ * in cents. `null` = no cap (the default): the agent can spend up to the
2533
+ * workspace's serverless balance. Once reached, the agent's model calls
2534
+ * are refused until next month and its runs pause with that reason.
2535
+ */
2536
+ monthly_spend_cap_cents: number | null;
2530
2537
  }
2531
2538
  export interface LoopAgentStatus {
2532
2539
  /** An agent can exist in this environment at all. */
@@ -2665,8 +2672,16 @@ export interface LoopCandidate {
2665
2672
  metadata: Record<string, unknown>;
2666
2673
  created_at: string;
2667
2674
  }
2668
- /** The deterministic checks a grader can perform. */
2669
- export type LoopGraderKind = 'exact_match' | 'contains' | 'regex' | 'json_valid' | 'numeric' | 'tool_called';
2675
+ /**
2676
+ * The deterministic checks a grader can perform.
2677
+ *
2678
+ * `matches_gold` is the one kind with no expected value of its own: it
2679
+ * compares each answer to the gold answer recorded on that same conversation
2680
+ * (or the human correction when there is no gold), scores sampled
2681
+ * alternatives against the same gold, and is NOT APPLIED to a conversation
2682
+ * that carries neither, so it never marks down an unlabelled answer.
2683
+ */
2684
+ export type LoopGraderKind = 'matches_gold' | 'exact_match' | 'contains' | 'regex' | 'json_valid' | 'numeric' | 'tool_called';
2670
2685
  export interface LoopGraderConfig {
2671
2686
  expected?: string;
2672
2687
  /** Every phrase that must appear. */
@@ -2676,7 +2691,10 @@ export interface LoopGraderConfig {
2676
2691
  pattern?: string;
2677
2692
  /** Keys the answer must carry, for `json_valid`. */
2678
2693
  keys?: string[];
2679
- /** How far from `expected` still counts, for `numeric`. */
2694
+ /**
2695
+ * How far from `expected` still counts, for `numeric`. For `matches_gold`,
2696
+ * how far from the gold answer still counts when both are bare numbers.
2697
+ */
2680
2698
  tolerance?: number;
2681
2699
  /** The function that must have been called, for `tool_called`. */
2682
2700
  function?: string;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.1-dev.133",
3
+ "version": "0.2.1-dev.135",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",