runbios-sdk 0.2.1-dev.140 → 0.2.1-dev.142
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -1
- package/dist/index.js +1 -1
- package/dist/resources/loop.d.ts +26 -1
- package/dist/resources/loop.js +31 -0
- package/dist/types.d.ts +41 -1
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export declare const VERSION = "0.2.1-dev.
|
|
39
|
+
export declare const VERSION = "0.2.1-dev.142";
|
|
40
40
|
export declare class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
readonly models: Models;
|
package/dist/index.js
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export const VERSION = '0.2.1-dev.
|
|
39
|
+
export const VERSION = '0.2.1-dev.142';
|
|
40
40
|
export class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
models;
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { HttpClient } from '../client.js';
|
|
2
|
-
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, AgentSettings, AgentSettingsRequest } from '../types.js';
|
|
2
|
+
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, AgentSettings, AgentSettingsRequest } from '../types.js';
|
|
3
3
|
/**
|
|
4
4
|
* The Conscious Loop -- capture what your model was asked and answered, record
|
|
5
5
|
* whether it was right, and turn those judgements into training data.
|
|
@@ -398,6 +398,25 @@ export declare class Loop {
|
|
|
398
398
|
* nothing: ask again and the same items come back.
|
|
399
399
|
*/
|
|
400
400
|
takeWork(runId: string, limit?: number): Promise<LoopJudgeWork>;
|
|
401
|
+
/**
|
|
402
|
+
* What happened to each conversation in a run, and why.
|
|
403
|
+
*
|
|
404
|
+
* `takeWork` hands out what is still PENDING, so a finished run answers it
|
|
405
|
+
* with an empty list. This answers with every item and the outcome on it:
|
|
406
|
+
* `status`, `scored_at`, and `error` — the reason the caller gave for an
|
|
407
|
+
* item it could not score, which is where a wrong model slug or a refused
|
|
408
|
+
* key actually shows up. A run that ends "scored 0, failed 3" is otherwise a
|
|
409
|
+
* number with no detail behind it.
|
|
410
|
+
*
|
|
411
|
+
* Pass `status: 'failed'` for the usual question. At most 500 items come
|
|
412
|
+
* back at a time; when `has_more` is true, call again with the `next_offset`
|
|
413
|
+
* from the reply.
|
|
414
|
+
*/
|
|
415
|
+
listRunItems(runId: string, opts?: {
|
|
416
|
+
status?: LoopJudgeRunItemStatus;
|
|
417
|
+
limit?: number;
|
|
418
|
+
offset?: number;
|
|
419
|
+
}): Promise<LoopJudgeRunItems>;
|
|
401
420
|
/**
|
|
402
421
|
* Hand the scores back. A verdict for a conversation outside this run, or
|
|
403
422
|
* scoring something the rubric never asked for, is refused and reported in
|
|
@@ -420,6 +439,12 @@ export declare class Loop {
|
|
|
420
439
|
* automatic judges and sample runs make model calls billed to the
|
|
421
440
|
* workspace. Idempotent.
|
|
422
441
|
*
|
|
442
|
+
* A judge whose model the serving gateway will not route is NOT resumed:
|
|
443
|
+
* making it automatic would buy a run that fails every conversation. It
|
|
444
|
+
* stays paused, `judges_still_paused` counts those, and each one carries
|
|
445
|
+
* `auto_pause_reason` saying so. Point it at a model that is served and
|
|
446
|
+
* turn the agent on again.
|
|
447
|
+
*
|
|
423
448
|
* `monthly_spend_cap_cents` caps what that key may spend on model calls in
|
|
424
449
|
* a calendar month. Omitted = no cap on a fresh key, and an existing cap is
|
|
425
450
|
* left as it is; sent while the agent is already on, it moves the cap on
|
package/dist/resources/loop.js
CHANGED
|
@@ -559,6 +559,31 @@ export class Loop {
|
|
|
559
559
|
async takeWork(runId, limit = 20) {
|
|
560
560
|
return this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}/work?limit=${limit}`);
|
|
561
561
|
}
|
|
562
|
+
/**
|
|
563
|
+
* What happened to each conversation in a run, and why.
|
|
564
|
+
*
|
|
565
|
+
* `takeWork` hands out what is still PENDING, so a finished run answers it
|
|
566
|
+
* with an empty list. This answers with every item and the outcome on it:
|
|
567
|
+
* `status`, `scored_at`, and `error` — the reason the caller gave for an
|
|
568
|
+
* item it could not score, which is where a wrong model slug or a refused
|
|
569
|
+
* key actually shows up. A run that ends "scored 0, failed 3" is otherwise a
|
|
570
|
+
* number with no detail behind it.
|
|
571
|
+
*
|
|
572
|
+
* Pass `status: 'failed'` for the usual question. At most 500 items come
|
|
573
|
+
* back at a time; when `has_more` is true, call again with the `next_offset`
|
|
574
|
+
* from the reply.
|
|
575
|
+
*/
|
|
576
|
+
async listRunItems(runId, opts = {}) {
|
|
577
|
+
const qs = new URLSearchParams();
|
|
578
|
+
if (opts.status)
|
|
579
|
+
qs.set('status', opts.status);
|
|
580
|
+
if (opts.limit != null)
|
|
581
|
+
qs.set('limit', String(opts.limit));
|
|
582
|
+
if (opts.offset)
|
|
583
|
+
qs.set('offset', String(opts.offset));
|
|
584
|
+
const suffix = qs.toString() ? `?${qs.toString()}` : '';
|
|
585
|
+
return this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}/items${suffix}`);
|
|
586
|
+
}
|
|
562
587
|
/**
|
|
563
588
|
* Hand the scores back. A verdict for a conversation outside this run, or
|
|
564
589
|
* scoring something the rubric never asked for, is refused and reported in
|
|
@@ -591,6 +616,12 @@ export class Loop {
|
|
|
591
616
|
* automatic judges and sample runs make model calls billed to the
|
|
592
617
|
* workspace. Idempotent.
|
|
593
618
|
*
|
|
619
|
+
* A judge whose model the serving gateway will not route is NOT resumed:
|
|
620
|
+
* making it automatic would buy a run that fails every conversation. It
|
|
621
|
+
* stays paused, `judges_still_paused` counts those, and each one carries
|
|
622
|
+
* `auto_pause_reason` saying so. Point it at a model that is served and
|
|
623
|
+
* turn the agent on again.
|
|
624
|
+
*
|
|
594
625
|
* `monthly_spend_cap_cents` caps what that key may spend on model calls in
|
|
595
626
|
* a calendar month. Omitted = no cap on a fresh key, and an existing cap is
|
|
596
627
|
* left as it is; sent while the agent is already on, it moves the cap on
|
package/dist/types.d.ts
CHANGED
|
@@ -2486,6 +2486,19 @@ export interface LoopJudge {
|
|
|
2486
2486
|
enabled: boolean;
|
|
2487
2487
|
/** The platform's agent runs this judge, on the workspace's own serverless account. */
|
|
2488
2488
|
auto: boolean;
|
|
2489
|
+
/**
|
|
2490
|
+
* Was automatic when the agent was turned off. Turning the agent back on
|
|
2491
|
+
* restores exactly these judges.
|
|
2492
|
+
*/
|
|
2493
|
+
auto_paused: boolean;
|
|
2494
|
+
/**
|
|
2495
|
+
* Why the last turn-on did NOT restore this judge. Null is the ordinary
|
|
2496
|
+
* state. Set when the resume declined to make it automatic because the model
|
|
2497
|
+
* it names is not one the serving gateway will route — handing it back would
|
|
2498
|
+
* buy a run that fails every conversation. It stays paused, so pointing it at
|
|
2499
|
+
* a model that routes and turning the agent on again brings it back.
|
|
2500
|
+
*/
|
|
2501
|
+
auto_pause_reason: string | null;
|
|
2489
2502
|
created_by: string | null;
|
|
2490
2503
|
created_at: string;
|
|
2491
2504
|
updated_at: string;
|
|
@@ -2517,7 +2530,13 @@ export interface LoopJudgeRun {
|
|
|
2517
2530
|
id: string;
|
|
2518
2531
|
judge_id: string;
|
|
2519
2532
|
judge_name?: string;
|
|
2520
|
-
|
|
2533
|
+
/**
|
|
2534
|
+
* `abandoned` is a run you opened and never drained: nothing was scored on
|
|
2535
|
+
* it for a week and items were still waiting. Nothing is deleted and posting
|
|
2536
|
+
* verdicts to it still works and still closes it as done. It exists so that
|
|
2537
|
+
* `open` keeps meaning "somebody is working on this".
|
|
2538
|
+
*/
|
|
2539
|
+
status: 'open' | 'done' | 'abandoned';
|
|
2521
2540
|
instructions: string;
|
|
2522
2541
|
dimensions: LoopJudgeDimension[];
|
|
2523
2542
|
model: string | null;
|
|
@@ -2630,6 +2649,27 @@ export interface LoopJudgeWork {
|
|
|
2630
2649
|
items: LoopJudgeWorkItem[];
|
|
2631
2650
|
remaining: number;
|
|
2632
2651
|
}
|
|
2652
|
+
export type LoopJudgeRunItemStatus = 'pending' | 'scored' | 'failed';
|
|
2653
|
+
/** What happened to one conversation in one run. */
|
|
2654
|
+
export interface LoopJudgeRunItem {
|
|
2655
|
+
trace_id: string;
|
|
2656
|
+
status: LoopJudgeRunItemStatus;
|
|
2657
|
+
/**
|
|
2658
|
+
* Why this one could not be scored, as the caller reported it. This is where
|
|
2659
|
+
* a wrong model name or a refused key shows up; null for anything that is
|
|
2660
|
+
* not failed.
|
|
2661
|
+
*/
|
|
2662
|
+
error: string | null;
|
|
2663
|
+
scored_at: string | null;
|
|
2664
|
+
}
|
|
2665
|
+
export interface LoopJudgeRunItems {
|
|
2666
|
+
run: LoopJudgeRun;
|
|
2667
|
+
items: LoopJudgeRunItem[];
|
|
2668
|
+
/** True when the page ended before the run did. */
|
|
2669
|
+
has_more: boolean;
|
|
2670
|
+
/** Where to carry on from. Present only when `has_more`. */
|
|
2671
|
+
next_offset?: number;
|
|
2672
|
+
}
|
|
2633
2673
|
/** One scored conversation going back. */
|
|
2634
2674
|
export interface LoopJudgeVerdict {
|
|
2635
2675
|
trace_id: string;
|