runbios-sdk 0.2.1-dev.123 → 0.2.1-dev.128
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -1
- package/dist/index.js +1 -1
- package/dist/resources/loop.d.ts +38 -1
- package/dist/resources/loop.js +55 -0
- package/dist/resources/training.js +4 -0
- package/dist/types.d.ts +105 -0
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export declare const VERSION = "0.2.1-dev.
|
|
39
|
+
export declare const VERSION = "0.2.1-dev.128";
|
|
40
40
|
export declare class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
readonly models: Models;
|
package/dist/index.js
CHANGED
|
@@ -36,7 +36,7 @@ import { Loop } from './resources/loop.js';
|
|
|
36
36
|
* SDK version. Sent as part of the User-Agent header.
|
|
37
37
|
* Must match package.json "version" -- enforced by a contract test.
|
|
38
38
|
*/
|
|
39
|
-
export const VERSION = '0.2.1-dev.
|
|
39
|
+
export const VERSION = '0.2.1-dev.128';
|
|
40
40
|
export class RunBiOS {
|
|
41
41
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
42
42
|
models;
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { HttpClient } from '../client.js';
|
|
2
|
-
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopStats, LoopJudge, LoopJudgeParams, LoopJudgeRun, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult } from '../types.js';
|
|
2
|
+
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult } from '../types.js';
|
|
3
3
|
/**
|
|
4
4
|
* The Conscious Loop -- capture what your model was asked and answered, record
|
|
5
5
|
* whether it was right, and turn those judgements into training data.
|
|
@@ -401,4 +401,41 @@ export declare class Loop {
|
|
|
401
401
|
* abandoned rather than completed.
|
|
402
402
|
*/
|
|
403
403
|
postVerdicts(runId: string, verdicts: LoopJudgeVerdict[], finish?: boolean): Promise<LoopJudgeVerdictResult>;
|
|
404
|
+
/**
|
|
405
|
+
* Can an agent run here, is one running, and is it on for this workspace.
|
|
406
|
+
* `available` is about the environment; `online` about the worker;
|
|
407
|
+
* `credential` about this workspace.
|
|
408
|
+
*/
|
|
409
|
+
agentStatus(): Promise<LoopAgentStatus>;
|
|
410
|
+
/**
|
|
411
|
+
* Turn the agent on: mints the workspace's managed serverless key. After
|
|
412
|
+
* this, automatic judges and sample runs make model calls billed to the
|
|
413
|
+
* workspace. Idempotent.
|
|
414
|
+
*/
|
|
415
|
+
enableAgent(): Promise<LoopAgentCredential>;
|
|
416
|
+
/**
|
|
417
|
+
* Turn the agent off: revokes its key and sets every automatic judge back to
|
|
418
|
+
* manual. Open runs stop where they are and continue if it is turned back
|
|
419
|
+
* on. Nothing already scored or written is removed.
|
|
420
|
+
*/
|
|
421
|
+
disableAgent(): Promise<{
|
|
422
|
+
revoked: boolean;
|
|
423
|
+
judges_paused: number;
|
|
424
|
+
message: string;
|
|
425
|
+
}>;
|
|
426
|
+
/**
|
|
427
|
+
* Ask the agent to write `n` alternative answers to each conversation in a
|
|
428
|
+
* slice with `model`, score each with `judge_id`, and store them as
|
|
429
|
+
* candidates. This is how preference pairs are made without a person
|
|
430
|
+
* writing each one: the curation pass pairs the best sample against the
|
|
431
|
+
* worst wherever the gap is real.
|
|
432
|
+
*
|
|
433
|
+
* Cost: up to `n` calls to write plus `n` to judge, per conversation, at the
|
|
434
|
+
* workspace's serverless rate; `selection.sample` caps the conversations
|
|
435
|
+
* (max 200) and `n` is capped at 8. Opening a run turns the agent on if it
|
|
436
|
+
* is off.
|
|
437
|
+
*/
|
|
438
|
+
createSampleRun(params: LoopSampleRunParams): Promise<LoopSampleRun>;
|
|
439
|
+
listSampleRuns(): Promise<LoopSampleRun[]>;
|
|
440
|
+
getSampleRun(runId: string): Promise<LoopSampleRun>;
|
|
404
441
|
}
|
package/dist/resources/loop.js
CHANGED
|
@@ -564,4 +564,59 @@ export class Loop {
|
|
|
564
564
|
async postVerdicts(runId, verdicts, finish = false) {
|
|
565
565
|
return this._http.fetchPost(`/api/loop/runs/${encodeURIComponent(runId)}/verdicts`, { verdicts, finish });
|
|
566
566
|
}
|
|
567
|
+
// ── the agent ─────────────────────────────────────────────────────────
|
|
568
|
+
//
|
|
569
|
+
// The worker that calls a model on the workspace's behalf: it applies
|
|
570
|
+
// automatic judges and writes sample answers. It spends through a managed
|
|
571
|
+
// serverless key OF THIS WORKSPACE, so every call is billed exactly like
|
|
572
|
+
// one of your own, and turning it off revokes that key.
|
|
573
|
+
/**
|
|
574
|
+
* Can an agent run here, is one running, and is it on for this workspace.
|
|
575
|
+
* `available` is about the environment; `online` about the worker;
|
|
576
|
+
* `credential` about this workspace.
|
|
577
|
+
*/
|
|
578
|
+
async agentStatus() {
|
|
579
|
+
return this._http.fetchGet('/api/loop/agent');
|
|
580
|
+
}
|
|
581
|
+
/**
|
|
582
|
+
* Turn the agent on: mints the workspace's managed serverless key. After
|
|
583
|
+
* this, automatic judges and sample runs make model calls billed to the
|
|
584
|
+
* workspace. Idempotent.
|
|
585
|
+
*/
|
|
586
|
+
async enableAgent() {
|
|
587
|
+
const res = await this._http.fetchPost('/api/loop/agent', {});
|
|
588
|
+
return res.credential;
|
|
589
|
+
}
|
|
590
|
+
/**
|
|
591
|
+
* Turn the agent off: revokes its key and sets every automatic judge back to
|
|
592
|
+
* manual. Open runs stop where they are and continue if it is turned back
|
|
593
|
+
* on. Nothing already scored or written is removed.
|
|
594
|
+
*/
|
|
595
|
+
async disableAgent() {
|
|
596
|
+
return this._http.fetchDelete('/api/loop/agent');
|
|
597
|
+
}
|
|
598
|
+
/**
|
|
599
|
+
* Ask the agent to write `n` alternative answers to each conversation in a
|
|
600
|
+
* slice with `model`, score each with `judge_id`, and store them as
|
|
601
|
+
* candidates. This is how preference pairs are made without a person
|
|
602
|
+
* writing each one: the curation pass pairs the best sample against the
|
|
603
|
+
* worst wherever the gap is real.
|
|
604
|
+
*
|
|
605
|
+
* Cost: up to `n` calls to write plus `n` to judge, per conversation, at the
|
|
606
|
+
* workspace's serverless rate; `selection.sample` caps the conversations
|
|
607
|
+
* (max 200) and `n` is capped at 8. Opening a run turns the agent on if it
|
|
608
|
+
* is off.
|
|
609
|
+
*/
|
|
610
|
+
async createSampleRun(params) {
|
|
611
|
+
const res = await this._http.fetchPost('/api/loop/sample-runs', params);
|
|
612
|
+
return res.run;
|
|
613
|
+
}
|
|
614
|
+
async listSampleRuns() {
|
|
615
|
+
const res = await this._http.fetchGet('/api/loop/sample-runs');
|
|
616
|
+
return res.runs;
|
|
617
|
+
}
|
|
618
|
+
async getSampleRun(runId) {
|
|
619
|
+
const res = await this._http.fetchGet(`/api/loop/sample-runs/${encodeURIComponent(runId)}`);
|
|
620
|
+
return res.run;
|
|
621
|
+
}
|
|
567
622
|
}
|
|
@@ -69,6 +69,10 @@ function buildTrainingRequest(params) {
|
|
|
69
69
|
body.network_volume_id = params.networkVolumeId;
|
|
70
70
|
if (params.datasetSampleLimits !== undefined)
|
|
71
71
|
body.dataset_sample_limits = params.datasetSampleLimits;
|
|
72
|
+
if (params.datasetSampling !== undefined)
|
|
73
|
+
body.dataset_sampling = params.datasetSampling;
|
|
74
|
+
if (params.datasetSamplingStrategy !== undefined)
|
|
75
|
+
body.dataset_sampling_strategy = params.datasetSamplingStrategy;
|
|
72
76
|
if (params.datasetMixing !== undefined)
|
|
73
77
|
body.dataset_mixing = params.datasetMixing;
|
|
74
78
|
if (params.mixing !== undefined)
|
package/dist/types.d.ts
CHANGED
|
@@ -548,6 +548,21 @@ export interface TrainingCreateParams {
|
|
|
548
548
|
* large dataset without importing a trimmed copy.
|
|
549
549
|
*/
|
|
550
550
|
datasetSampleLimits?: Record<string, number>;
|
|
551
|
+
/**
|
|
552
|
+
* Per-dataset sampling (key = a dataset ID from datasetIds): exactly one of
|
|
553
|
+
* `rows` or `percent` (of the dataset's usable rows for this method). Asking
|
|
554
|
+
* for more than the dataset holds uses every usable row and the run says
|
|
555
|
+
* so. Resolved to a row count identically by preflight and create, recorded
|
|
556
|
+
* in the mix manifest, and reproduced exactly on resume.
|
|
557
|
+
*/
|
|
558
|
+
datasetSampling?: Record<string, DatasetSampling>;
|
|
559
|
+
/**
|
|
560
|
+
* How sampled rows are chosen: `first` (default) keeps the first N usable
|
|
561
|
+
* rows in file order -- what a curriculum wants; `random` draws a seeded
|
|
562
|
+
* uniform sample of the usable rows (kept in file order, seed = the mixing
|
|
563
|
+
* plan's seed, so a resume draws the same rows).
|
|
564
|
+
*/
|
|
565
|
+
datasetSamplingStrategy?: 'first' | 'random';
|
|
551
566
|
/** Training method. */
|
|
552
567
|
method: TrainingMethod;
|
|
553
568
|
/** Adapter type. */
|
|
@@ -1211,6 +1226,12 @@ export interface GPUPricingResponse {
|
|
|
1211
1226
|
stale?: boolean;
|
|
1212
1227
|
stale_message?: string;
|
|
1213
1228
|
}
|
|
1229
|
+
/** One dataset's sampling request: exactly one of rows or percent. */
|
|
1230
|
+
export interface DatasetSampling {
|
|
1231
|
+
rows?: number;
|
|
1232
|
+
/** Percent (0-100) of the dataset's usable rows for the training method. */
|
|
1233
|
+
percent?: number;
|
|
1234
|
+
}
|
|
1214
1235
|
/** Parameters understood by the authenticated training GPU-options endpoint. */
|
|
1215
1236
|
export interface GPUOptionsParams {
|
|
1216
1237
|
modelId: string;
|
|
@@ -2439,6 +2460,8 @@ export interface LoopJudge {
|
|
|
2439
2460
|
model: string | null;
|
|
2440
2461
|
write_gold: boolean;
|
|
2441
2462
|
enabled: boolean;
|
|
2463
|
+
/** The platform's agent runs this judge, on the workspace's own serverless account. */
|
|
2464
|
+
auto: boolean;
|
|
2442
2465
|
created_by: string | null;
|
|
2443
2466
|
created_at: string;
|
|
2444
2467
|
updated_at: string;
|
|
@@ -2456,6 +2479,14 @@ export interface LoopJudgeParams {
|
|
|
2456
2479
|
*/
|
|
2457
2480
|
write_gold?: boolean;
|
|
2458
2481
|
enabled?: boolean;
|
|
2482
|
+
/**
|
|
2483
|
+
* Hand the running of this judge to the platform's agent: new conversations
|
|
2484
|
+
* in its slice are scored as they arrive, one model call each, on THIS
|
|
2485
|
+
* WORKSPACE'S OWN serverless account (the agent spends through a managed key
|
|
2486
|
+
* of yours, "Conscious Loop" in your key list). Requires `model`. Off, you
|
|
2487
|
+
* run the model yourself with startRun / takeWork / postVerdicts.
|
|
2488
|
+
*/
|
|
2489
|
+
auto?: boolean;
|
|
2459
2490
|
}
|
|
2460
2491
|
/** One pass of one rubric over one slice. */
|
|
2461
2492
|
export interface LoopJudgeRun {
|
|
@@ -2469,6 +2500,80 @@ export interface LoopJudgeRun {
|
|
|
2469
2500
|
selected: number;
|
|
2470
2501
|
scored: number;
|
|
2471
2502
|
failed: number;
|
|
2503
|
+
/** Who drains it: your own code ('caller') or the platform's agent ('platform'). */
|
|
2504
|
+
runner: 'caller' | 'platform';
|
|
2505
|
+
/** The agent's last complaint about this run, or null while it is working. */
|
|
2506
|
+
last_error: string | null;
|
|
2507
|
+
last_activity_at: string | null;
|
|
2508
|
+
created_by: string | null;
|
|
2509
|
+
created_at: string;
|
|
2510
|
+
finished_at: string | null;
|
|
2511
|
+
}
|
|
2512
|
+
/** The workspace's managed serverless key the agent spends through. Never the secret. */
|
|
2513
|
+
export interface LoopAgentCredential {
|
|
2514
|
+
workspace_id: string;
|
|
2515
|
+
key_id: string;
|
|
2516
|
+
key_prefix: string;
|
|
2517
|
+
created_by: string | null;
|
|
2518
|
+
created_at: string;
|
|
2519
|
+
updated_at: string;
|
|
2520
|
+
last_used_at: string | null;
|
|
2521
|
+
last_error: string | null;
|
|
2522
|
+
revoked_at: string | null;
|
|
2523
|
+
}
|
|
2524
|
+
export interface LoopAgentStatus {
|
|
2525
|
+
/** An agent can exist in this environment at all. */
|
|
2526
|
+
available: boolean;
|
|
2527
|
+
/** Which piece is missing when it cannot. */
|
|
2528
|
+
reason: string;
|
|
2529
|
+
/** A worker has checked in within the last minute. */
|
|
2530
|
+
online: boolean;
|
|
2531
|
+
agent: {
|
|
2532
|
+
seen_at: string;
|
|
2533
|
+
passes: number;
|
|
2534
|
+
judge_items: number;
|
|
2535
|
+
samples: number;
|
|
2536
|
+
} | null;
|
|
2537
|
+
credential: LoopAgentCredential | null;
|
|
2538
|
+
open_judge_runs: number;
|
|
2539
|
+
open_sample_runs: number;
|
|
2540
|
+
auto_judges: number;
|
|
2541
|
+
}
|
|
2542
|
+
export interface LoopSampleSelection extends LoopJudgeSelection {
|
|
2543
|
+
/** Skip conversations that already have alternatives. */
|
|
2544
|
+
only_unsampled?: boolean;
|
|
2545
|
+
}
|
|
2546
|
+
export interface LoopSampleRunParams {
|
|
2547
|
+
/** The model that writes the alternatives. For distillation, the teacher. */
|
|
2548
|
+
model: string;
|
|
2549
|
+
/** Alternatives per conversation, 1-8. Default 4. */
|
|
2550
|
+
n?: number;
|
|
2551
|
+
/** 0-2. Default 0.8; 0 makes every sample the same answer. */
|
|
2552
|
+
temperature?: number;
|
|
2553
|
+
/** 16-8192. Default 1024. */
|
|
2554
|
+
max_tokens?: number;
|
|
2555
|
+
/** Score each sample with this judge as it is written. */
|
|
2556
|
+
judge_id?: string;
|
|
2557
|
+
selection?: LoopSampleSelection;
|
|
2558
|
+
}
|
|
2559
|
+
/** "Write N alternatives to each conversation in this slice, and score them." */
|
|
2560
|
+
export interface LoopSampleRun {
|
|
2561
|
+
id: string;
|
|
2562
|
+
workspace_id: string;
|
|
2563
|
+
model: string;
|
|
2564
|
+
n: number;
|
|
2565
|
+
temperature: number;
|
|
2566
|
+
max_tokens: number;
|
|
2567
|
+
judge_id: string | null;
|
|
2568
|
+
judge_name: string | null;
|
|
2569
|
+
selection: LoopSampleSelection;
|
|
2570
|
+
status: 'open' | 'done';
|
|
2571
|
+
selected: number;
|
|
2572
|
+
done: number;
|
|
2573
|
+
failed: number;
|
|
2574
|
+
samples: number;
|
|
2575
|
+
last_error: string | null;
|
|
2576
|
+
last_activity_at: string | null;
|
|
2472
2577
|
created_by: string | null;
|
|
2473
2578
|
created_at: string;
|
|
2474
2579
|
finished_at: string | null;
|