runbios-sdk 0.2.1-dev.141 → 0.2.1-dev.142

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export declare const VERSION = "0.2.1-dev.141";
39
+ export declare const VERSION = "0.2.1-dev.142";
40
40
  export declare class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  readonly models: Models;
package/dist/index.js CHANGED
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
36
36
  * SDK version. Sent as part of the User-Agent header.
37
37
  * Must match package.json "version" -- enforced by a contract test.
38
38
  */
39
- export const VERSION = '0.2.1-dev.141';
39
+ export const VERSION = '0.2.1-dev.142';
40
40
  export class RunBiOS {
41
41
  /** Search models, fetch configs, check adapter compatibility. */
42
42
  models;
@@ -1,5 +1,5 @@
1
1
  import type { HttpClient } from '../client.js';
2
- import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, AgentSettings, AgentSettingsRequest } from '../types.js';
2
+ import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, AgentSettings, AgentSettingsRequest } from '../types.js';
3
3
  /**
4
4
  * The Conscious Loop -- capture what your model was asked and answered, record
5
5
  * whether it was right, and turn those judgements into training data.
@@ -398,6 +398,25 @@ export declare class Loop {
398
398
  * nothing: ask again and the same items come back.
399
399
  */
400
400
  takeWork(runId: string, limit?: number): Promise<LoopJudgeWork>;
401
+ /**
402
+ * What happened to each conversation in a run, and why.
403
+ *
404
+ * `takeWork` hands out what is still PENDING, so a finished run answers it
405
+ * with an empty list. This answers with every item and the outcome on it:
406
+ * `status`, `scored_at`, and `error` — the reason the caller gave for an
407
+ * item it could not score, which is where a wrong model slug or a refused
408
+ * key actually shows up. A run that ends "scored 0, failed 3" is otherwise a
409
+ * number with no detail behind it.
410
+ *
411
+ * Pass `status: 'failed'` for the usual question. At most 500 items come
412
+ * back at a time; when `has_more` is true, call again with the `next_offset`
413
+ * from the reply.
414
+ */
415
+ listRunItems(runId: string, opts?: {
416
+ status?: LoopJudgeRunItemStatus;
417
+ limit?: number;
418
+ offset?: number;
419
+ }): Promise<LoopJudgeRunItems>;
401
420
  /**
402
421
  * Hand the scores back. A verdict for a conversation outside this run, or
403
422
  * scoring something the rubric never asked for, is refused and reported in
@@ -420,6 +439,12 @@ export declare class Loop {
420
439
  * automatic judges and sample runs make model calls billed to the
421
440
  * workspace. Idempotent.
422
441
  *
442
+ * A judge whose model the serving gateway will not route is NOT resumed:
443
+ * making it automatic would buy a run that fails every conversation. It
444
+ * stays paused, `judges_still_paused` counts those, and each one carries
445
+ * `auto_pause_reason` saying so. Point it at a model that is served and
446
+ * turn the agent on again.
447
+ *
423
448
  * `monthly_spend_cap_cents` caps what that key may spend on model calls in
424
449
  * a calendar month. Omitted = no cap on a fresh key, and an existing cap is
425
450
  * left as it is; sent while the agent is already on, it moves the cap on
@@ -559,6 +559,31 @@ export class Loop {
559
559
  async takeWork(runId, limit = 20) {
560
560
  return this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}/work?limit=${limit}`);
561
561
  }
562
+ /**
563
+ * What happened to each conversation in a run, and why.
564
+ *
565
+ * `takeWork` hands out what is still PENDING, so a finished run answers it
566
+ * with an empty list. This answers with every item and the outcome on it:
567
+ * `status`, `scored_at`, and `error` — the reason the caller gave for an
568
+ * item it could not score, which is where a wrong model slug or a refused
569
+ * key actually shows up. A run that ends "scored 0, failed 3" is otherwise a
570
+ * number with no detail behind it.
571
+ *
572
+ * Pass `status: 'failed'` for the usual question. At most 500 items come
573
+ * back at a time; when `has_more` is true, call again with the `next_offset`
574
+ * from the reply.
575
+ */
576
+ async listRunItems(runId, opts = {}) {
577
+ const qs = new URLSearchParams();
578
+ if (opts.status)
579
+ qs.set('status', opts.status);
580
+ if (opts.limit != null)
581
+ qs.set('limit', String(opts.limit));
582
+ if (opts.offset)
583
+ qs.set('offset', String(opts.offset));
584
+ const suffix = qs.toString() ? `?${qs.toString()}` : '';
585
+ return this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}/items${suffix}`);
586
+ }
562
587
  /**
563
588
  * Hand the scores back. A verdict for a conversation outside this run, or
564
589
  * scoring something the rubric never asked for, is refused and reported in
@@ -591,6 +616,12 @@ export class Loop {
591
616
  * automatic judges and sample runs make model calls billed to the
592
617
  * workspace. Idempotent.
593
618
  *
619
+ * A judge whose model the serving gateway will not route is NOT resumed:
620
+ * making it automatic would buy a run that fails every conversation. It
621
+ * stays paused, `judges_still_paused` counts those, and each one carries
622
+ * `auto_pause_reason` saying so. Point it at a model that is served and
623
+ * turn the agent on again.
624
+ *
594
625
  * `monthly_spend_cap_cents` caps what that key may spend on model calls in
595
626
  * a calendar month. Omitted = no cap on a fresh key, and an existing cap is
596
627
  * left as it is; sent while the agent is already on, it moves the cap on
package/dist/types.d.ts CHANGED
@@ -2486,6 +2486,19 @@ export interface LoopJudge {
2486
2486
  enabled: boolean;
2487
2487
  /** The platform's agent runs this judge, on the workspace's own serverless account. */
2488
2488
  auto: boolean;
2489
+ /**
2490
+ * Was automatic when the agent was turned off. Turning the agent back on
2491
+ * restores exactly these judges.
2492
+ */
2493
+ auto_paused: boolean;
2494
+ /**
2495
+ * Why the last turn-on did NOT restore this judge. Null is the ordinary
2496
+ * state. Set when the resume declined to make it automatic because the model
2497
+ * it names is not one the serving gateway will route — handing it back would
2498
+ * buy a run that fails every conversation. It stays paused, so pointing it at
2499
+ * a model that routes and turning the agent on again brings it back.
2500
+ */
2501
+ auto_pause_reason: string | null;
2489
2502
  created_by: string | null;
2490
2503
  created_at: string;
2491
2504
  updated_at: string;
@@ -2517,7 +2530,13 @@ export interface LoopJudgeRun {
2517
2530
  id: string;
2518
2531
  judge_id: string;
2519
2532
  judge_name?: string;
2520
- status: 'open' | 'done';
2533
+ /**
2534
+ * `abandoned` is a run you opened and never drained: nothing was scored on
2535
+ * it for a week and items were still waiting. Nothing is deleted and posting
2536
+ * verdicts to it still works and still closes it as done. It exists so that
2537
+ * `open` keeps meaning "somebody is working on this".
2538
+ */
2539
+ status: 'open' | 'done' | 'abandoned';
2521
2540
  instructions: string;
2522
2541
  dimensions: LoopJudgeDimension[];
2523
2542
  model: string | null;
@@ -2630,6 +2649,27 @@ export interface LoopJudgeWork {
2630
2649
  items: LoopJudgeWorkItem[];
2631
2650
  remaining: number;
2632
2651
  }
2652
+ export type LoopJudgeRunItemStatus = 'pending' | 'scored' | 'failed';
2653
+ /** What happened to one conversation in one run. */
2654
+ export interface LoopJudgeRunItem {
2655
+ trace_id: string;
2656
+ status: LoopJudgeRunItemStatus;
2657
+ /**
2658
+ * Why this one could not be scored, as the caller reported it. This is where
2659
+ * a wrong model name or a refused key shows up; null for anything that is
2660
+ * not failed.
2661
+ */
2662
+ error: string | null;
2663
+ scored_at: string | null;
2664
+ }
2665
+ export interface LoopJudgeRunItems {
2666
+ run: LoopJudgeRun;
2667
+ items: LoopJudgeRunItem[];
2668
+ /** True when the page ended before the run did. */
2669
+ has_more: boolean;
2670
+ /** Where to carry on from. Present only when `has_more`. */
2671
+ next_offset?: number;
2672
+ }
2633
2673
  /** One scored conversation going back. */
2634
2674
  export interface LoopJudgeVerdict {
2635
2675
  trace_id: string;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.1-dev.141",
3
+ "version": "0.2.1-dev.142",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",