runbios-sdk 0.2.1-dev.141 → 0.2.1-dev.145
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -1
- package/dist/index.js +1 -1
- package/dist/resources/loop.d.ts +26 -1
- package/dist/resources/loop.js +31 -0
- package/dist/types.d.ts +46 -2
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export declare const VERSION = "0.2.1-dev.
|
|
39
|
+
export declare const VERSION = "0.2.1-dev.145";
|
|
40
40
|
export declare class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
readonly models: Models;
|
package/dist/index.js
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export const VERSION = '0.2.1-dev.
|
|
39
|
+
export const VERSION = '0.2.1-dev.145';
|
|
40
40
|
export class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
models;
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { HttpClient } from '../client.js';
|
|
2
|
-
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, AgentSettings, AgentSettingsRequest } from '../types.js';
|
|
2
|
+
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, AgentSettings, AgentSettingsRequest } from '../types.js';
|
|
3
3
|
/**
|
|
4
4
|
* The Conscious Loop -- capture what your model was asked and answered, record
|
|
5
5
|
* whether it was right, and turn those judgements into training data.
|
|
@@ -398,6 +398,25 @@ export declare class Loop {
|
|
|
398
398
|
* nothing: ask again and the same items come back.
|
|
399
399
|
*/
|
|
400
400
|
takeWork(runId: string, limit?: number): Promise<LoopJudgeWork>;
|
|
401
|
+
/**
|
|
402
|
+
* What happened to each conversation in a run, and why.
|
|
403
|
+
*
|
|
404
|
+
* `takeWork` hands out what is still PENDING, so a finished run answers it
|
|
405
|
+
* with an empty list. This answers with every item and the outcome on it:
|
|
406
|
+
* `status`, `scored_at`, and `error` — the reason the caller gave for an
|
|
407
|
+
* item it could not score, which is where a wrong model slug or a refused
|
|
408
|
+
* key actually shows up. A run that ends "scored 0, failed 3" is otherwise a
|
|
409
|
+
* number with no detail behind it.
|
|
410
|
+
*
|
|
411
|
+
* Pass `status: 'failed'` for the usual question. At most 500 items come
|
|
412
|
+
* back at a time; when `has_more` is true, call again with the `next_offset`
|
|
413
|
+
* from the reply.
|
|
414
|
+
*/
|
|
415
|
+
listRunItems(runId: string, opts?: {
|
|
416
|
+
status?: LoopJudgeRunItemStatus;
|
|
417
|
+
limit?: number;
|
|
418
|
+
offset?: number;
|
|
419
|
+
}): Promise<LoopJudgeRunItems>;
|
|
401
420
|
/**
|
|
402
421
|
* Hand the scores back. A verdict for a conversation outside this run, or
|
|
403
422
|
* scoring something the rubric never asked for, is refused and reported in
|
|
@@ -420,6 +439,12 @@ export declare class Loop {
|
|
|
420
439
|
* automatic judges and sample runs make model calls billed to the
|
|
421
440
|
* workspace. Idempotent.
|
|
422
441
|
*
|
|
442
|
+
* A judge whose model the serving gateway will not route is NOT resumed:
|
|
443
|
+
* making it automatic would buy a run that fails every conversation. It
|
|
444
|
+
* stays paused, `judges_still_paused` counts those, and each one carries
|
|
445
|
+
* `auto_pause_reason` saying so. Point it at a model that is served and
|
|
446
|
+
* turn the agent on again.
|
|
447
|
+
*
|
|
423
448
|
* `monthly_spend_cap_cents` caps what that key may spend on model calls in
|
|
424
449
|
* a calendar month. Omitted = no cap on a fresh key, and an existing cap is
|
|
425
450
|
* left as it is; sent while the agent is already on, it moves the cap on
|
package/dist/resources/loop.js
CHANGED
|
@@ -559,6 +559,31 @@ export class Loop {
|
|
|
559
559
|
async takeWork(runId, limit = 20) {
|
|
560
560
|
return this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}/work?limit=${limit}`);
|
|
561
561
|
}
|
|
562
|
+
/**
|
|
563
|
+
* What happened to each conversation in a run, and why.
|
|
564
|
+
*
|
|
565
|
+
* `takeWork` hands out what is still PENDING, so a finished run answers it
|
|
566
|
+
* with an empty list. This answers with every item and the outcome on it:
|
|
567
|
+
* `status`, `scored_at`, and `error` — the reason the caller gave for an
|
|
568
|
+
* item it could not score, which is where a wrong model slug or a refused
|
|
569
|
+
* key actually shows up. A run that ends "scored 0, failed 3" is otherwise a
|
|
570
|
+
* number with no detail behind it.
|
|
571
|
+
*
|
|
572
|
+
* Pass `status: 'failed'` for the usual question. At most 500 items come
|
|
573
|
+
* back at a time; when `has_more` is true, call again with the `next_offset`
|
|
574
|
+
* from the reply.
|
|
575
|
+
*/
|
|
576
|
+
async listRunItems(runId, opts = {}) {
|
|
577
|
+
const qs = new URLSearchParams();
|
|
578
|
+
if (opts.status)
|
|
579
|
+
qs.set('status', opts.status);
|
|
580
|
+
if (opts.limit != null)
|
|
581
|
+
qs.set('limit', String(opts.limit));
|
|
582
|
+
if (opts.offset)
|
|
583
|
+
qs.set('offset', String(opts.offset));
|
|
584
|
+
const suffix = qs.toString() ? `?${qs.toString()}` : '';
|
|
585
|
+
return this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}/items${suffix}`);
|
|
586
|
+
}
|
|
562
587
|
/**
|
|
563
588
|
* Hand the scores back. A verdict for a conversation outside this run, or
|
|
564
589
|
* scoring something the rubric never asked for, is refused and reported in
|
|
@@ -591,6 +616,12 @@ export class Loop {
|
|
|
591
616
|
* automatic judges and sample runs make model calls billed to the
|
|
592
617
|
* workspace. Idempotent.
|
|
593
618
|
*
|
|
619
|
+
* A judge whose model the serving gateway will not route is NOT resumed:
|
|
620
|
+
* making it automatic would buy a run that fails every conversation. It
|
|
621
|
+
* stays paused, `judges_still_paused` counts those, and each one carries
|
|
622
|
+
* `auto_pause_reason` saying so. Point it at a model that is served and
|
|
623
|
+
* turn the agent on again.
|
|
624
|
+
*
|
|
594
625
|
* `monthly_spend_cap_cents` caps what that key may spend on model calls in
|
|
595
626
|
* a calendar month. Omitted = no cap on a fresh key, and an existing cap is
|
|
596
627
|
* left as it is; sent while the agent is already on, it moves the cap on
|
package/dist/types.d.ts
CHANGED
|
@@ -2350,7 +2350,11 @@ export interface LoopDatasetCreateParams {
|
|
|
2350
2350
|
to?: string;
|
|
2351
2351
|
/** Narrow to one slice. A label with children selects them too. */
|
|
2352
2352
|
label?: string;
|
|
2353
|
-
/**
|
|
2353
|
+
/**
|
|
2354
|
+
* Take this many of what the filters matched. Which ones is arbitrary but
|
|
2355
|
+
* REPEATABLE: the same conversations always give the same slice, so two
|
|
2356
|
+
* builds of one spec describe the same set.
|
|
2357
|
+
*/
|
|
2354
2358
|
sample?: number;
|
|
2355
2359
|
/** Holds back your most recent work, not a random slice. 0-50. */
|
|
2356
2360
|
holdout_percent?: number;
|
|
@@ -2486,6 +2490,19 @@ export interface LoopJudge {
|
|
|
2486
2490
|
enabled: boolean;
|
|
2487
2491
|
/** The platform's agent runs this judge, on the workspace's own serverless account. */
|
|
2488
2492
|
auto: boolean;
|
|
2493
|
+
/**
|
|
2494
|
+
* Was automatic when the agent was turned off. Turning the agent back on
|
|
2495
|
+
* restores exactly these judges.
|
|
2496
|
+
*/
|
|
2497
|
+
auto_paused: boolean;
|
|
2498
|
+
/**
|
|
2499
|
+
* Why the last turn-on did NOT restore this judge. Null is the ordinary
|
|
2500
|
+
* state. Set when the resume declined to make it automatic because the model
|
|
2501
|
+
* it names is not one the serving gateway will route — handing it back would
|
|
2502
|
+
* buy a run that fails every conversation. It stays paused, so pointing it at
|
|
2503
|
+
* a model that routes and turning the agent on again brings it back.
|
|
2504
|
+
*/
|
|
2505
|
+
auto_pause_reason: string | null;
|
|
2489
2506
|
created_by: string | null;
|
|
2490
2507
|
created_at: string;
|
|
2491
2508
|
updated_at: string;
|
|
@@ -2517,7 +2534,13 @@ export interface LoopJudgeRun {
|
|
|
2517
2534
|
id: string;
|
|
2518
2535
|
judge_id: string;
|
|
2519
2536
|
judge_name?: string;
|
|
2520
|
-
|
|
2537
|
+
/**
|
|
2538
|
+
* `abandoned` is a run you opened and never drained: nothing was scored on
|
|
2539
|
+
* it for a week and items were still waiting. Nothing is deleted and posting
|
|
2540
|
+
* verdicts to it still works and still closes it as done. It exists so that
|
|
2541
|
+
* `open` keeps meaning "somebody is working on this".
|
|
2542
|
+
*/
|
|
2543
|
+
status: 'open' | 'done' | 'abandoned';
|
|
2521
2544
|
instructions: string;
|
|
2522
2545
|
dimensions: LoopJudgeDimension[];
|
|
2523
2546
|
model: string | null;
|
|
@@ -2630,6 +2653,27 @@ export interface LoopJudgeWork {
|
|
|
2630
2653
|
items: LoopJudgeWorkItem[];
|
|
2631
2654
|
remaining: number;
|
|
2632
2655
|
}
|
|
2656
|
+
export type LoopJudgeRunItemStatus = 'pending' | 'scored' | 'failed';
|
|
2657
|
+
/** What happened to one conversation in one run. */
|
|
2658
|
+
export interface LoopJudgeRunItem {
|
|
2659
|
+
trace_id: string;
|
|
2660
|
+
status: LoopJudgeRunItemStatus;
|
|
2661
|
+
/**
|
|
2662
|
+
* Why this one could not be scored, as the caller reported it. This is where
|
|
2663
|
+
* a wrong model name or a refused key shows up; null for anything that is
|
|
2664
|
+
* not failed.
|
|
2665
|
+
*/
|
|
2666
|
+
error: string | null;
|
|
2667
|
+
scored_at: string | null;
|
|
2668
|
+
}
|
|
2669
|
+
export interface LoopJudgeRunItems {
|
|
2670
|
+
run: LoopJudgeRun;
|
|
2671
|
+
items: LoopJudgeRunItem[];
|
|
2672
|
+
/** True when the page ended before the run did. */
|
|
2673
|
+
has_more: boolean;
|
|
2674
|
+
/** Where to carry on from. Present only when `has_more`. */
|
|
2675
|
+
next_offset?: number;
|
|
2676
|
+
}
|
|
2633
2677
|
/** One scored conversation going back. */
|
|
2634
2678
|
export interface LoopJudgeVerdict {
|
|
2635
2679
|
trace_id: string;
|