runbios-sdk 0.2.1-dev.105 → 0.2.1-dev.107
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -1
- package/dist/index.js +1 -1
- package/dist/resources/loop.d.ts +75 -3
- package/dist/resources/loop.js +89 -2
- package/dist/types.d.ts +89 -0
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
|
|
|
35
35
|
* SDK version. Sent as part of the User-Agent header.
|
|
36
36
|
* Must match package.json "version" -- enforced by a contract test.
|
|
37
37
|
*/
|
|
38
|
-
export declare const VERSION = "0.2.1-dev.
|
|
38
|
+
export declare const VERSION = "0.2.1-dev.107";
|
|
39
39
|
export declare class RunBiOS {
|
|
40
40
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
41
41
|
readonly models: Models;
|
package/dist/index.js
CHANGED
|
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
|
|
|
35
35
|
* SDK version. Sent as part of the User-Agent header.
|
|
36
36
|
* Must match package.json "version" -- enforced by a contract test.
|
|
37
37
|
*/
|
|
38
|
-
export const VERSION = '0.2.1-dev.
|
|
38
|
+
export const VERSION = '0.2.1-dev.107';
|
|
39
39
|
export class RunBiOS {
|
|
40
40
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
41
41
|
models;
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { HttpClient } from '../client.js';
|
|
2
|
-
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats } from '../types.js';
|
|
2
|
+
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats } from '../types.js';
|
|
3
3
|
/**
|
|
4
4
|
* The Conscious Loop -- capture what your model was asked and answered, record
|
|
5
5
|
* whether it was right, and turn those judgements into training data.
|
|
@@ -18,7 +18,7 @@ import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCapture
|
|
|
18
18
|
* await client.loop.setConfig('my-agent', { enabled: true, retention_days: 30 });
|
|
19
19
|
*
|
|
20
20
|
* // 2. Send what your model was asked and what it answered. Works whoever
|
|
21
|
-
* // served it -- us,
|
|
21
|
+
* // served it -- us, another provider, or your own servers.
|
|
22
22
|
* const { trace_id } = await client.loop.capture({
|
|
23
23
|
* deployment_id: 'my-agent',
|
|
24
24
|
* model: 'gpt-4o',
|
|
@@ -46,7 +46,7 @@ export declare class Loop {
|
|
|
46
46
|
* Send one exchange to be captured.
|
|
47
47
|
*
|
|
48
48
|
* Works regardless of who served the model: ours, OpenAI, Fireworks, your own
|
|
49
|
-
*
|
|
49
|
+
* own servers, an agent framework. Pass `tool_calls` and `tools` when the turn
|
|
50
50
|
* invoked a function -- without them a tool-using exchange trains the model to
|
|
51
51
|
* answer in prose where it should have called something.
|
|
52
52
|
*
|
|
@@ -90,6 +90,37 @@ export declare class Loop {
|
|
|
90
90
|
signal(traceId: string, params: LoopSignalParams): Promise<LoopSignal>;
|
|
91
91
|
/** Every verdict on one conversation, oldest first. */
|
|
92
92
|
listSignals(traceId: string): Promise<LoopSignal[]>;
|
|
93
|
+
/**
|
|
94
|
+
* Submit an alternative answer to a prompt already captured.
|
|
95
|
+
*
|
|
96
|
+
* This is what makes preference learning scale. A human rewrite is the best
|
|
97
|
+
* signal there is and the least available -- somebody has to sit down and
|
|
98
|
+
* write it. Sample the model several times for the same prompt, score the
|
|
99
|
+
* samples, and a DPO pair falls out automatically: best becomes chosen,
|
|
100
|
+
* worst becomes rejected. Point a stronger model at the prompt instead and
|
|
101
|
+
* the same machinery does distillation.
|
|
102
|
+
*
|
|
103
|
+
* Score them, or they cannot pair: an unscored alternative says nothing about
|
|
104
|
+
* which answer anybody prefers. `score_source` is required alongside a score,
|
|
105
|
+
* because precedence between a verifier, a judge and a person is the whole
|
|
106
|
+
* reason we record who judged.
|
|
107
|
+
*
|
|
108
|
+
* A human correction always outranks any score.
|
|
109
|
+
*
|
|
110
|
+
* @example
|
|
111
|
+
* ```ts
|
|
112
|
+
* for (const sample of await sampleMyModel(prompt, 4)) {
|
|
113
|
+
* await client.loop.addCandidate(traceId, {
|
|
114
|
+
* completion: sample.text,
|
|
115
|
+
* score: await myVerifier(sample.text),
|
|
116
|
+
* score_source: 'verifier',
|
|
117
|
+
* });
|
|
118
|
+
* }
|
|
119
|
+
* ```
|
|
120
|
+
*/
|
|
121
|
+
addCandidate(traceId: string, params: LoopCandidateParams): Promise<LoopCandidate>;
|
|
122
|
+
/** Every alternative answer recorded for one prompt, oldest first. */
|
|
123
|
+
listCandidates(traceId: string): Promise<LoopCandidate[]>;
|
|
93
124
|
/**
|
|
94
125
|
* Build a training set from the feedback recorded so far.
|
|
95
126
|
*
|
|
@@ -141,6 +172,47 @@ export declare class Loop {
|
|
|
141
172
|
deleted: boolean;
|
|
142
173
|
dataset_id: string;
|
|
143
174
|
}>;
|
|
175
|
+
/**
|
|
176
|
+
* Write a rule that scores answers without a person.
|
|
177
|
+
*
|
|
178
|
+
* Human review is the most trustworthy feedback and the least available. A
|
|
179
|
+
* grader is written once and applied to every answer afterwards: did it
|
|
180
|
+
* contain the required phrase, did it parse as the schema you asked for, is
|
|
181
|
+
* the number within tolerance of the known answer, did it call the function
|
|
182
|
+
* it should have.
|
|
183
|
+
*
|
|
184
|
+
* Every kind here is DETERMINISTIC -- no model call, no network. That is why
|
|
185
|
+
* a verifier outranks a judge when they disagree: it cannot be flattered and
|
|
186
|
+
* it cannot drift between runs.
|
|
187
|
+
*
|
|
188
|
+
* `weight` is the multiplier: a criterion that matters twice as much gets
|
|
189
|
+
* twice the weight, and the combined score is the weighted mean over the
|
|
190
|
+
* rules that actually applied.
|
|
191
|
+
*
|
|
192
|
+
* A caution worth knowing before you write a set: a rule made only of
|
|
193
|
+
* `forbidden` phrases is satisfied VACUOUSLY by an answer that says nothing.
|
|
194
|
+
* Pair it with a `required` phrase, or you are rewarding silence.
|
|
195
|
+
*/
|
|
196
|
+
createGrader(params: LoopGraderParams): Promise<LoopGrader>;
|
|
197
|
+
/** Every rule this workspace has written. */
|
|
198
|
+
listGraders(): Promise<LoopGrader[]>;
|
|
199
|
+
deleteGrader(id: string): Promise<{
|
|
200
|
+
deleted: boolean;
|
|
201
|
+
grader_id: string;
|
|
202
|
+
}>;
|
|
203
|
+
/**
|
|
204
|
+
* Apply this workspace's rules to one captured answer AND to every
|
|
205
|
+
* alternative sampled for it.
|
|
206
|
+
*
|
|
207
|
+
* Grading both in one pass is the point: scoring only the original gives a
|
|
208
|
+
* verdict, while scoring the samples as well gives the preference pair. Sample
|
|
209
|
+
* your model, call this, and you have DPO data with nobody reading anything.
|
|
210
|
+
*
|
|
211
|
+
* A run where no rule could apply writes NOTHING -- no verdict, no scores.
|
|
212
|
+
* Recording a zero that no rule produced would poison curation with a
|
|
213
|
+
* judgement nobody reached, so `applied: 0` is reported instead.
|
|
214
|
+
*/
|
|
215
|
+
grade(traceId: string): Promise<LoopGradeResult>;
|
|
144
216
|
/** Every source this workspace has configured. */
|
|
145
217
|
listConfigs(): Promise<LoopConfig[]>;
|
|
146
218
|
/**
|
package/dist/resources/loop.js
CHANGED
|
@@ -16,7 +16,7 @@
|
|
|
16
16
|
* await client.loop.setConfig('my-agent', { enabled: true, retention_days: 30 });
|
|
17
17
|
*
|
|
18
18
|
* // 2. Send what your model was asked and what it answered. Works whoever
|
|
19
|
-
* // served it -- us,
|
|
19
|
+
* // served it -- us, another provider, or your own servers.
|
|
20
20
|
* const { trace_id } = await client.loop.capture({
|
|
21
21
|
* deployment_id: 'my-agent',
|
|
22
22
|
* model: 'gpt-4o',
|
|
@@ -47,7 +47,7 @@ export class Loop {
|
|
|
47
47
|
* Send one exchange to be captured.
|
|
48
48
|
*
|
|
49
49
|
* Works regardless of who served the model: ours, OpenAI, Fireworks, your own
|
|
50
|
-
*
|
|
50
|
+
* own servers, an agent framework. Pass `tool_calls` and `tools` when the turn
|
|
51
51
|
* invoked a function -- without them a tool-using exchange trains the model to
|
|
52
52
|
* answer in prose where it should have called something.
|
|
53
53
|
*
|
|
@@ -121,6 +121,44 @@ export class Loop {
|
|
|
121
121
|
const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`);
|
|
122
122
|
return res.signals;
|
|
123
123
|
}
|
|
124
|
+
// ── alternative answers ───────────────────────────────────────────────
|
|
125
|
+
/**
|
|
126
|
+
* Submit an alternative answer to a prompt already captured.
|
|
127
|
+
*
|
|
128
|
+
* This is what makes preference learning scale. A human rewrite is the best
|
|
129
|
+
* signal there is and the least available -- somebody has to sit down and
|
|
130
|
+
* write it. Sample the model several times for the same prompt, score the
|
|
131
|
+
* samples, and a DPO pair falls out automatically: best becomes chosen,
|
|
132
|
+
* worst becomes rejected. Point a stronger model at the prompt instead and
|
|
133
|
+
* the same machinery does distillation.
|
|
134
|
+
*
|
|
135
|
+
* Score them, or they cannot pair: an unscored alternative says nothing about
|
|
136
|
+
* which answer anybody prefers. `score_source` is required alongside a score,
|
|
137
|
+
* because precedence between a verifier, a judge and a person is the whole
|
|
138
|
+
* reason we record who judged.
|
|
139
|
+
*
|
|
140
|
+
* A human correction always outranks any score.
|
|
141
|
+
*
|
|
142
|
+
* @example
|
|
143
|
+
* ```ts
|
|
144
|
+
* for (const sample of await sampleMyModel(prompt, 4)) {
|
|
145
|
+
* await client.loop.addCandidate(traceId, {
|
|
146
|
+
* completion: sample.text,
|
|
147
|
+
* score: await myVerifier(sample.text),
|
|
148
|
+
* score_source: 'verifier',
|
|
149
|
+
* });
|
|
150
|
+
* }
|
|
151
|
+
* ```
|
|
152
|
+
*/
|
|
153
|
+
async addCandidate(traceId, params) {
|
|
154
|
+
const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/candidates`, params);
|
|
155
|
+
return res.candidate;
|
|
156
|
+
}
|
|
157
|
+
/** Every alternative answer recorded for one prompt, oldest first. */
|
|
158
|
+
async listCandidates(traceId) {
|
|
159
|
+
const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/candidates`);
|
|
160
|
+
return res.candidates;
|
|
161
|
+
}
|
|
124
162
|
// ── training sets ─────────────────────────────────────────────────────
|
|
125
163
|
/**
|
|
126
164
|
* Build a training set from the feedback recorded so far.
|
|
@@ -198,6 +236,55 @@ export class Loop {
|
|
|
198
236
|
async deleteDataset(id) {
|
|
199
237
|
return this._http.fetchDelete(`/api/loop/datasets/${encodeURIComponent(id)}`);
|
|
200
238
|
}
|
|
239
|
+
// ── graders ───────────────────────────────────────────────────────────
|
|
240
|
+
/**
|
|
241
|
+
* Write a rule that scores answers without a person.
|
|
242
|
+
*
|
|
243
|
+
* Human review is the most trustworthy feedback and the least available. A
|
|
244
|
+
* grader is written once and applied to every answer afterwards: did it
|
|
245
|
+
* contain the required phrase, did it parse as the schema you asked for, is
|
|
246
|
+
* the number within tolerance of the known answer, did it call the function
|
|
247
|
+
* it should have.
|
|
248
|
+
*
|
|
249
|
+
* Every kind here is DETERMINISTIC -- no model call, no network. That is why
|
|
250
|
+
* a verifier outranks a judge when they disagree: it cannot be flattered and
|
|
251
|
+
* it cannot drift between runs.
|
|
252
|
+
*
|
|
253
|
+
* `weight` is the multiplier: a criterion that matters twice as much gets
|
|
254
|
+
* twice the weight, and the combined score is the weighted mean over the
|
|
255
|
+
* rules that actually applied.
|
|
256
|
+
*
|
|
257
|
+
* A caution worth knowing before you write a set: a rule made only of
|
|
258
|
+
* `forbidden` phrases is satisfied VACUOUSLY by an answer that says nothing.
|
|
259
|
+
* Pair it with a `required` phrase, or you are rewarding silence.
|
|
260
|
+
*/
|
|
261
|
+
async createGrader(params) {
|
|
262
|
+
const res = await this._http.fetchPost('/api/loop/graders', params);
|
|
263
|
+
return res.grader;
|
|
264
|
+
}
|
|
265
|
+
/** Every rule this workspace has written. */
|
|
266
|
+
async listGraders() {
|
|
267
|
+
const res = await this._http.fetchGet('/api/loop/graders');
|
|
268
|
+
return res.graders;
|
|
269
|
+
}
|
|
270
|
+
async deleteGrader(id) {
|
|
271
|
+
return this._http.fetchDelete(`/api/loop/graders/${encodeURIComponent(id)}`);
|
|
272
|
+
}
|
|
273
|
+
/**
|
|
274
|
+
* Apply this workspace's rules to one captured answer AND to every
|
|
275
|
+
* alternative sampled for it.
|
|
276
|
+
*
|
|
277
|
+
* Grading both in one pass is the point: scoring only the original gives a
|
|
278
|
+
* verdict, while scoring the samples as well gives the preference pair. Sample
|
|
279
|
+
* your model, call this, and you have DPO data with nobody reading anything.
|
|
280
|
+
*
|
|
281
|
+
* A run where no rule could apply writes NOTHING -- no verdict, no scores.
|
|
282
|
+
* Recording a zero that no rule produced would poison curation with a
|
|
283
|
+
* judgement nobody reached, so `applied: 0` is reported instead.
|
|
284
|
+
*/
|
|
285
|
+
async grade(traceId) {
|
|
286
|
+
return this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/grade`);
|
|
287
|
+
}
|
|
201
288
|
// ── capture settings ──────────────────────────────────────────────────
|
|
202
289
|
/** Every source this workspace has configured. */
|
|
203
290
|
async listConfigs() {
|
package/dist/types.d.ts
CHANGED
|
@@ -2140,3 +2140,92 @@ export interface LoopStats {
|
|
|
2140
2140
|
/** Upper bounds: deduplication runs when a set is built. */
|
|
2141
2141
|
ready: Record<LoopMethod, number>;
|
|
2142
2142
|
}
|
|
2143
|
+
export interface LoopCandidateParams {
|
|
2144
|
+
completion?: string;
|
|
2145
|
+
/** An alternative that is itself a tool call. */
|
|
2146
|
+
tool_calls?: LoopToolCall[];
|
|
2147
|
+
/** Which model produced this alternative. The teacher, when distilling. */
|
|
2148
|
+
model?: string;
|
|
2149
|
+
model_version?: string;
|
|
2150
|
+
/** Unscored alternatives cannot pair -- a missing score is not a low one. */
|
|
2151
|
+
score?: number;
|
|
2152
|
+
/** Required alongside a score: human | verifier | judge | behavioural. */
|
|
2153
|
+
score_source?: LoopSource;
|
|
2154
|
+
reason?: string;
|
|
2155
|
+
metadata?: Record<string, unknown>;
|
|
2156
|
+
}
|
|
2157
|
+
export interface LoopCandidate {
|
|
2158
|
+
id: string;
|
|
2159
|
+
trace_id: string;
|
|
2160
|
+
model: string | null;
|
|
2161
|
+
model_version: string | null;
|
|
2162
|
+
completion: string;
|
|
2163
|
+
tool_calls?: LoopToolCall[];
|
|
2164
|
+
score: number | null;
|
|
2165
|
+
score_source: LoopSource | null;
|
|
2166
|
+
reason: string | null;
|
|
2167
|
+
metadata: Record<string, unknown>;
|
|
2168
|
+
created_at: string;
|
|
2169
|
+
}
|
|
2170
|
+
/** The deterministic checks a grader can perform. */
|
|
2171
|
+
export type LoopGraderKind = 'exact_match' | 'contains' | 'regex' | 'json_valid' | 'numeric' | 'tool_called';
|
|
2172
|
+
export interface LoopGraderConfig {
|
|
2173
|
+
expected?: string;
|
|
2174
|
+
/** Every phrase that must appear. */
|
|
2175
|
+
required?: string[];
|
|
2176
|
+
/** Phrases that must not. On their own these are passed by silence. */
|
|
2177
|
+
forbidden?: string[];
|
|
2178
|
+
pattern?: string;
|
|
2179
|
+
/** Keys the answer must carry, for `json_valid`. */
|
|
2180
|
+
keys?: string[];
|
|
2181
|
+
/** How far from `expected` still counts, for `numeric`. */
|
|
2182
|
+
tolerance?: number;
|
|
2183
|
+
/** The function that must have been called, for `tool_called`. */
|
|
2184
|
+
function?: string;
|
|
2185
|
+
case_sensitive?: boolean;
|
|
2186
|
+
}
|
|
2187
|
+
export interface LoopGraderParams {
|
|
2188
|
+
name: string;
|
|
2189
|
+
kind: LoopGraderKind;
|
|
2190
|
+
config: LoopGraderConfig;
|
|
2191
|
+
/** The multiplier. Twice as important, twice the weight. Must exceed zero. */
|
|
2192
|
+
weight?: number;
|
|
2193
|
+
enabled?: boolean;
|
|
2194
|
+
/** Limit the rule to one capture source. Omit to apply it everywhere. */
|
|
2195
|
+
deployment_id?: string;
|
|
2196
|
+
}
|
|
2197
|
+
export interface LoopGrader extends LoopGraderParams {
|
|
2198
|
+
id: string;
|
|
2199
|
+
workspace_id: string;
|
|
2200
|
+
weight: number;
|
|
2201
|
+
enabled: boolean;
|
|
2202
|
+
created_by: string | null;
|
|
2203
|
+
created_at: string;
|
|
2204
|
+
updated_at: string;
|
|
2205
|
+
}
|
|
2206
|
+
export interface LoopGraderResult {
|
|
2207
|
+
grader_id: string;
|
|
2208
|
+
name: string;
|
|
2209
|
+
kind: string;
|
|
2210
|
+
weight: number;
|
|
2211
|
+
passed: boolean;
|
|
2212
|
+
score: number;
|
|
2213
|
+
/** What happened, in words -- "failed" alone sends you to read the rule. */
|
|
2214
|
+
detail: string;
|
|
2215
|
+
/** The rule had no opinion. Excluded from the score rather than counted 0. */
|
|
2216
|
+
skipped: boolean;
|
|
2217
|
+
}
|
|
2218
|
+
export interface LoopGradeReport {
|
|
2219
|
+
/** Weighted mean over the rules that applied, 0 to 1. */
|
|
2220
|
+
score: number;
|
|
2221
|
+
results: LoopGraderResult[];
|
|
2222
|
+
/** The denominator. 1.0 from one rule is not 1.0 from six. */
|
|
2223
|
+
applied: number;
|
|
2224
|
+
skipped: number;
|
|
2225
|
+
}
|
|
2226
|
+
export interface LoopGradeResult {
|
|
2227
|
+
report: LoopGradeReport;
|
|
2228
|
+
/** Null when no rule applied, because no verdict was invented. */
|
|
2229
|
+
signal_id: string | null;
|
|
2230
|
+
candidates_graded: number;
|
|
2231
|
+
}
|