runbios-sdk 0.2.1-dev.104 → 0.2.1-dev.106
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +8 -1
- package/dist/index.js +9 -1
- package/dist/resources/loop.d.ts +200 -0
- package/dist/resources/loop.js +277 -0
- package/dist/types.d.ts +213 -1
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -30,11 +30,12 @@ import { Training } from './resources/training.js';
|
|
|
30
30
|
import { Wallet } from './resources/wallet.js';
|
|
31
31
|
import { GPU } from './resources/gpu.js';
|
|
32
32
|
import { Inference } from './resources/inference.js';
|
|
33
|
+
import { Loop } from './resources/loop.js';
|
|
33
34
|
/**
|
|
34
35
|
* SDK version. Sent as part of the User-Agent header.
|
|
35
36
|
* Must match package.json "version" -- enforced by a contract test.
|
|
36
37
|
*/
|
|
37
|
-
export declare const VERSION = "0.2.1-dev.
|
|
38
|
+
export declare const VERSION = "0.2.1-dev.106";
|
|
38
39
|
export declare class RunBiOS {
|
|
39
40
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
40
41
|
readonly models: Models;
|
|
@@ -51,6 +52,12 @@ export declare class RunBiOS {
|
|
|
51
52
|
* serving deployments, plus OpenAI-compatible non-streaming and SSE calls.
|
|
52
53
|
*/
|
|
53
54
|
readonly inference: Inference;
|
|
55
|
+
/**
|
|
56
|
+
* The Conscious Loop. Capture what your model was asked and answered, record
|
|
57
|
+
* whether it was right, and turn those judgements into training data for
|
|
58
|
+
* SFT, DPO or GRPO. Nothing is captured until you turn it on for a source.
|
|
59
|
+
*/
|
|
60
|
+
readonly loop: Loop;
|
|
54
61
|
/** @internal */
|
|
55
62
|
private readonly _http;
|
|
56
63
|
constructor(config: BiOSConfig);
|
package/dist/index.js
CHANGED
|
@@ -30,11 +30,12 @@ import { Training } from './resources/training.js';
|
|
|
30
30
|
import { Wallet } from './resources/wallet.js';
|
|
31
31
|
import { GPU } from './resources/gpu.js';
|
|
32
32
|
import { Inference } from './resources/inference.js';
|
|
33
|
+
import { Loop } from './resources/loop.js';
|
|
33
34
|
/**
|
|
34
35
|
* SDK version. Sent as part of the User-Agent header.
|
|
35
36
|
* Must match package.json "version" -- enforced by a contract test.
|
|
36
37
|
*/
|
|
37
|
-
export const VERSION = '0.2.1-dev.
|
|
38
|
+
export const VERSION = '0.2.1-dev.106';
|
|
38
39
|
export class RunBiOS {
|
|
39
40
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
40
41
|
models;
|
|
@@ -51,6 +52,12 @@ export class RunBiOS {
|
|
|
51
52
|
* serving deployments, plus OpenAI-compatible non-streaming and SSE calls.
|
|
52
53
|
*/
|
|
53
54
|
inference;
|
|
55
|
+
/**
|
|
56
|
+
* The Conscious Loop. Capture what your model was asked and answered, record
|
|
57
|
+
* whether it was right, and turn those judgements into training data for
|
|
58
|
+
* SFT, DPO or GRPO. Nothing is captured until you turn it on for a source.
|
|
59
|
+
*/
|
|
60
|
+
loop;
|
|
54
61
|
/** @internal */
|
|
55
62
|
_http;
|
|
56
63
|
constructor(config) {
|
|
@@ -61,6 +68,7 @@ export class RunBiOS {
|
|
|
61
68
|
this.training = new Training(http);
|
|
62
69
|
this.wallet = new Wallet(http);
|
|
63
70
|
this.gpu = new GPU(http);
|
|
71
|
+
this.loop = new Loop(http);
|
|
64
72
|
// Inference-key resolution, most specific first:
|
|
65
73
|
// 1. an explicit inferenceKey (a per-deployment sk-bios-... key)
|
|
66
74
|
// 2. RUNBIOS_INFERENCE_KEY from the environment (legacy: BIOS_INFERENCE_KEY)
|
|
@@ -0,0 +1,200 @@
|
|
|
1
|
+
import type { HttpClient } from '../client.js';
|
|
2
|
+
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats } from '../types.js';
|
|
3
|
+
/**
|
|
4
|
+
* The Conscious Loop -- capture what your model was asked and answered, record
|
|
5
|
+
* whether it was right, and turn those judgements into training data.
|
|
6
|
+
*
|
|
7
|
+
* Nothing is captured until you turn it on for a source, and you can turn it
|
|
8
|
+
* off again at any time. Anything that looks like a credential or a personal
|
|
9
|
+
* detail is removed before the record is written, never afterwards.
|
|
10
|
+
*
|
|
11
|
+
* Requires an API key carrying `loop:read` and/or `loop:write`. Those scopes are
|
|
12
|
+
* deliberately absent from the Read Only preset, because what they return is
|
|
13
|
+
* your raw prompts and completions rather than catalog metadata.
|
|
14
|
+
*
|
|
15
|
+
* @example
|
|
16
|
+
* ```ts
|
|
17
|
+
* // 1. Decide that this source is recorded.
|
|
18
|
+
* await client.loop.setConfig('my-agent', { enabled: true, retention_days: 30 });
|
|
19
|
+
*
|
|
20
|
+
* // 2. Send what your model was asked and what it answered. Works whoever
|
|
21
|
+
* // served it -- us, another provider, or your own servers.
|
|
22
|
+
* const { trace_id } = await client.loop.capture({
|
|
23
|
+
* deployment_id: 'my-agent',
|
|
24
|
+
* model: 'gpt-4o',
|
|
25
|
+
* messages: [{ role: 'user', content: 'what is our refund window' }],
|
|
26
|
+
* completion: 'thirty days',
|
|
27
|
+
* });
|
|
28
|
+
*
|
|
29
|
+
* // 3. Say whether it was right. A rewrite is the most valuable answer here:
|
|
30
|
+
* // the model learns your version AND learns to avoid its own.
|
|
31
|
+
* await client.loop.signal(trace_id, {
|
|
32
|
+
* verdict: 'edited',
|
|
33
|
+
* correction: 'Thirty days from delivery, no questions asked.',
|
|
34
|
+
* });
|
|
35
|
+
*
|
|
36
|
+
* // 4. Turn the judgements into a training set and take the file.
|
|
37
|
+
* const ds = await client.loop.createDataset({ name: 'support', method: 'sft' });
|
|
38
|
+
* const jsonl = await client.loop.downloadDataset(ds.id);
|
|
39
|
+
* ```
|
|
40
|
+
*/
|
|
41
|
+
export declare class Loop {
|
|
42
|
+
private readonly _http;
|
|
43
|
+
/** @internal */
|
|
44
|
+
constructor(_http: HttpClient);
|
|
45
|
+
/**
|
|
46
|
+
* Send one exchange to be captured.
|
|
47
|
+
*
|
|
48
|
+
* Works regardless of who served the model: ours, OpenAI, Fireworks, your own
|
|
49
|
+
* own servers, an agent framework. Pass `tool_calls` and `tools` when the turn
|
|
50
|
+
* invoked a function -- without them a tool-using exchange trains the model to
|
|
51
|
+
* answer in prose where it should have called something.
|
|
52
|
+
*
|
|
53
|
+
* The returned `trace_id` is OURS. If you pass your own `request_id` it is
|
|
54
|
+
* kept as an idempotency handle: sending the same one again returns the same
|
|
55
|
+
* trace rather than storing a second copy, so a retry is safe.
|
|
56
|
+
*
|
|
57
|
+
* When the source is not enabled this resolves with `captured: false` and a
|
|
58
|
+
* reason instead of throwing -- capture must never be the thing that breaks
|
|
59
|
+
* your application.
|
|
60
|
+
*/
|
|
61
|
+
capture(params: LoopCaptureParams): Promise<LoopCaptureResult>;
|
|
62
|
+
/** List captured conversations, newest first. */
|
|
63
|
+
listTraces(params?: LoopTraceListParams): Promise<LoopTraceListResponse>;
|
|
64
|
+
/** Read one conversation, with every verdict recorded on it. */
|
|
65
|
+
getTrace(id: string): Promise<LoopTrace>;
|
|
66
|
+
/**
|
|
67
|
+
* Delete a conversation.
|
|
68
|
+
*
|
|
69
|
+
* It also leaves every training set built from it, including sets already
|
|
70
|
+
* exported -- so a customer asking you to delete a conversation genuinely
|
|
71
|
+
* removes it from what the model will learn next. Training runs that have
|
|
72
|
+
* already finished are not affected.
|
|
73
|
+
*/
|
|
74
|
+
deleteTrace(id: string): Promise<{
|
|
75
|
+
deleted: boolean;
|
|
76
|
+
trace_id: string;
|
|
77
|
+
}>;
|
|
78
|
+
/**
|
|
79
|
+
* Record a verdict on a conversation.
|
|
80
|
+
*
|
|
81
|
+
* Append-only: posting a second verdict does not replace the first. Two
|
|
82
|
+
* reviewers disagreeing about an answer is information worth keeping.
|
|
83
|
+
*
|
|
84
|
+
* `source` says who judged: `human`, `verifier` (a deterministic check --
|
|
85
|
+
* tests passed, schema valid), `judge` (a model grading a model), or
|
|
86
|
+
* `behavioural` (what the user did next). When several disagree, a human
|
|
87
|
+
* outranks a verifier, a verifier outranks a judge, and a judge outranks a
|
|
88
|
+
* behavioural hint.
|
|
89
|
+
*/
|
|
90
|
+
signal(traceId: string, params: LoopSignalParams): Promise<LoopSignal>;
|
|
91
|
+
/** Every verdict on one conversation, oldest first. */
|
|
92
|
+
listSignals(traceId: string): Promise<LoopSignal[]>;
|
|
93
|
+
/**
|
|
94
|
+
* Submit an alternative answer to a prompt already captured.
|
|
95
|
+
*
|
|
96
|
+
* This is what makes preference learning scale. A human rewrite is the best
|
|
97
|
+
* signal there is and the least available -- somebody has to sit down and
|
|
98
|
+
* write it. Sample the model several times for the same prompt, score the
|
|
99
|
+
* samples, and a DPO pair falls out automatically: best becomes chosen,
|
|
100
|
+
* worst becomes rejected. Point a stronger model at the prompt instead and
|
|
101
|
+
* the same machinery does distillation.
|
|
102
|
+
*
|
|
103
|
+
* Score them, or they cannot pair: an unscored alternative says nothing about
|
|
104
|
+
* which answer anybody prefers. `score_source` is required alongside a score,
|
|
105
|
+
* because precedence between a verifier, a judge and a person is the whole
|
|
106
|
+
* reason we record who judged.
|
|
107
|
+
*
|
|
108
|
+
* A human correction always outranks any score.
|
|
109
|
+
*
|
|
110
|
+
* @example
|
|
111
|
+
* ```ts
|
|
112
|
+
* for (const sample of await sampleMyModel(prompt, 4)) {
|
|
113
|
+
* await client.loop.addCandidate(traceId, {
|
|
114
|
+
* completion: sample.text,
|
|
115
|
+
* score: await myVerifier(sample.text),
|
|
116
|
+
* score_source: 'verifier',
|
|
117
|
+
* });
|
|
118
|
+
* }
|
|
119
|
+
* ```
|
|
120
|
+
*/
|
|
121
|
+
addCandidate(traceId: string, params: LoopCandidateParams): Promise<LoopCandidate>;
|
|
122
|
+
/** Every alternative answer recorded for one prompt, oldest first. */
|
|
123
|
+
listCandidates(traceId: string): Promise<LoopCandidate[]>;
|
|
124
|
+
/**
|
|
125
|
+
* Build a training set from the feedback recorded so far.
|
|
126
|
+
*
|
|
127
|
+
* The three methods need genuinely different things, so one set cannot be
|
|
128
|
+
* reshaped into another afterwards:
|
|
129
|
+
*
|
|
130
|
+
* - `sft` -- answers you marked right, and answers you rewrote.
|
|
131
|
+
* - `dpo` -- answers you REWROTE, so a better and a worse version of the same
|
|
132
|
+
* reply exist. Nothing else produces a pair.
|
|
133
|
+
* - `grpo` -- answers with a value or fact they can be checked against.
|
|
134
|
+
*
|
|
135
|
+
* `holdout_percent` holds back your most recent work rather than a random
|
|
136
|
+
* slice, so the evaluation measures whether the model generalised instead of
|
|
137
|
+
* memorised the same week.
|
|
138
|
+
*
|
|
139
|
+
* The result always reports `rejected_counts`: why rows were left out. A
|
|
140
|
+
* small set with a reason is useful; a small set without one is just alarming.
|
|
141
|
+
*/
|
|
142
|
+
createDataset(params: LoopDatasetCreateParams): Promise<LoopDataset>;
|
|
143
|
+
/** List training sets, newest first. */
|
|
144
|
+
listDatasets(params?: LoopDatasetListParams): Promise<LoopDataset[]>;
|
|
145
|
+
/** Read one training set and its curation report. */
|
|
146
|
+
getDataset(id: string): Promise<LoopDataset>;
|
|
147
|
+
/**
|
|
148
|
+
* Page through the rows a training set actually contains.
|
|
149
|
+
*
|
|
150
|
+
* Worth reading before you spend money on a run: each row carries the address
|
|
151
|
+
* of the conversation it was built from.
|
|
152
|
+
*/
|
|
153
|
+
listDatasetItems(id: string, params?: {
|
|
154
|
+
limit?: number;
|
|
155
|
+
offset?: number;
|
|
156
|
+
}): Promise<LoopDatasetItem[]>;
|
|
157
|
+
/**
|
|
158
|
+
* Download a training set as JSONL -- one training row per line, the format
|
|
159
|
+
* every trainer in this space reads.
|
|
160
|
+
*
|
|
161
|
+
* `split` defaults to the training rows; pass `holdout` for the slice held
|
|
162
|
+
* back, or `all` for both.
|
|
163
|
+
*/
|
|
164
|
+
downloadDataset(id: string, split?: 'train' | 'holdout' | 'all'): Promise<string>;
|
|
165
|
+
/**
|
|
166
|
+
* Delete a training set.
|
|
167
|
+
*
|
|
168
|
+
* The conversations it was built from are untouched -- a set is a selection,
|
|
169
|
+
* and discarding the selection must not discard the evidence.
|
|
170
|
+
*/
|
|
171
|
+
deleteDataset(id: string): Promise<{
|
|
172
|
+
deleted: boolean;
|
|
173
|
+
dataset_id: string;
|
|
174
|
+
}>;
|
|
175
|
+
/** Every source this workspace has configured. */
|
|
176
|
+
listConfigs(): Promise<LoopConfig[]>;
|
|
177
|
+
/**
|
|
178
|
+
* Read the capture setting for one source. A source nobody has configured
|
|
179
|
+
* reads back as disabled rather than missing, because "we are not recording
|
|
180
|
+
* this" is the honest answer.
|
|
181
|
+
*/
|
|
182
|
+
getConfig(source: string): Promise<LoopConfig>;
|
|
183
|
+
/**
|
|
184
|
+
* Turn capture on or off for a source, and choose how long it is kept.
|
|
185
|
+
*
|
|
186
|
+
* This is the consent decision the whole feature rests on: nothing is
|
|
187
|
+
* recorded until it is made, and the row remembers who made it and when.
|
|
188
|
+
* `sample_rate` below 1 records a deterministic fraction, chosen so that every
|
|
189
|
+
* turn of one conversation is captured or skipped together -- half a
|
|
190
|
+
* conversation makes training rows with holes in them.
|
|
191
|
+
*/
|
|
192
|
+
setConfig(source: string, params: LoopConfigParams): Promise<LoopConfig>;
|
|
193
|
+
/**
|
|
194
|
+
* How much there is, and how much of it each method could actually use.
|
|
195
|
+
*
|
|
196
|
+
* The `ready` numbers are upper bounds: duplicate questions are folded into
|
|
197
|
+
* one row while a set is built, so the finished count can be lower.
|
|
198
|
+
*/
|
|
199
|
+
stats(): Promise<LoopStats>;
|
|
200
|
+
}
|
|
@@ -0,0 +1,277 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The Conscious Loop -- capture what your model was asked and answered, record
|
|
3
|
+
* whether it was right, and turn those judgements into training data.
|
|
4
|
+
*
|
|
5
|
+
* Nothing is captured until you turn it on for a source, and you can turn it
|
|
6
|
+
* off again at any time. Anything that looks like a credential or a personal
|
|
7
|
+
* detail is removed before the record is written, never afterwards.
|
|
8
|
+
*
|
|
9
|
+
* Requires an API key carrying `loop:read` and/or `loop:write`. Those scopes are
|
|
10
|
+
* deliberately absent from the Read Only preset, because what they return is
|
|
11
|
+
* your raw prompts and completions rather than catalog metadata.
|
|
12
|
+
*
|
|
13
|
+
* @example
|
|
14
|
+
* ```ts
|
|
15
|
+
* // 1. Decide that this source is recorded.
|
|
16
|
+
* await client.loop.setConfig('my-agent', { enabled: true, retention_days: 30 });
|
|
17
|
+
*
|
|
18
|
+
* // 2. Send what your model was asked and what it answered. Works whoever
|
|
19
|
+
* // served it -- us, another provider, or your own servers.
|
|
20
|
+
* const { trace_id } = await client.loop.capture({
|
|
21
|
+
* deployment_id: 'my-agent',
|
|
22
|
+
* model: 'gpt-4o',
|
|
23
|
+
* messages: [{ role: 'user', content: 'what is our refund window' }],
|
|
24
|
+
* completion: 'thirty days',
|
|
25
|
+
* });
|
|
26
|
+
*
|
|
27
|
+
* // 3. Say whether it was right. A rewrite is the most valuable answer here:
|
|
28
|
+
* // the model learns your version AND learns to avoid its own.
|
|
29
|
+
* await client.loop.signal(trace_id, {
|
|
30
|
+
* verdict: 'edited',
|
|
31
|
+
* correction: 'Thirty days from delivery, no questions asked.',
|
|
32
|
+
* });
|
|
33
|
+
*
|
|
34
|
+
* // 4. Turn the judgements into a training set and take the file.
|
|
35
|
+
* const ds = await client.loop.createDataset({ name: 'support', method: 'sft' });
|
|
36
|
+
* const jsonl = await client.loop.downloadDataset(ds.id);
|
|
37
|
+
* ```
|
|
38
|
+
*/
|
|
39
|
+
export class Loop {
|
|
40
|
+
_http;
|
|
41
|
+
/** @internal */
|
|
42
|
+
constructor(_http) {
|
|
43
|
+
this._http = _http;
|
|
44
|
+
}
|
|
45
|
+
// ── capture ───────────────────────────────────────────────────────────
|
|
46
|
+
/**
|
|
47
|
+
* Send one exchange to be captured.
|
|
48
|
+
*
|
|
49
|
+
* Works regardless of who served the model: ours, OpenAI, Fireworks, your own
|
|
50
|
+
* own servers, an agent framework. Pass `tool_calls` and `tools` when the turn
|
|
51
|
+
* invoked a function -- without them a tool-using exchange trains the model to
|
|
52
|
+
* answer in prose where it should have called something.
|
|
53
|
+
*
|
|
54
|
+
* The returned `trace_id` is OURS. If you pass your own `request_id` it is
|
|
55
|
+
* kept as an idempotency handle: sending the same one again returns the same
|
|
56
|
+
* trace rather than storing a second copy, so a retry is safe.
|
|
57
|
+
*
|
|
58
|
+
* When the source is not enabled this resolves with `captured: false` and a
|
|
59
|
+
* reason instead of throwing -- capture must never be the thing that breaks
|
|
60
|
+
* your application.
|
|
61
|
+
*/
|
|
62
|
+
async capture(params) {
|
|
63
|
+
return this._http.fetchPost('/api/loop/traces', params);
|
|
64
|
+
}
|
|
65
|
+
// ── traces ────────────────────────────────────────────────────────────
|
|
66
|
+
/** List captured conversations, newest first. */
|
|
67
|
+
async listTraces(params = {}) {
|
|
68
|
+
const q = new URLSearchParams();
|
|
69
|
+
if (params.deployment_id)
|
|
70
|
+
q.set('deployment_id', params.deployment_id);
|
|
71
|
+
if (params.conversation_id)
|
|
72
|
+
q.set('conversation_id', params.conversation_id);
|
|
73
|
+
if (params.from)
|
|
74
|
+
q.set('from', params.from);
|
|
75
|
+
if (params.to)
|
|
76
|
+
q.set('to', params.to);
|
|
77
|
+
if (params.signalled)
|
|
78
|
+
q.set('signalled', 'true');
|
|
79
|
+
if (params.limit != null)
|
|
80
|
+
q.set('limit', String(params.limit));
|
|
81
|
+
if (params.offset != null)
|
|
82
|
+
q.set('offset', String(params.offset));
|
|
83
|
+
const qs = q.toString();
|
|
84
|
+
return this._http.fetchGet(`/api/loop/traces${qs ? `?${qs}` : ''}`);
|
|
85
|
+
}
|
|
86
|
+
/** Read one conversation, with every verdict recorded on it. */
|
|
87
|
+
async getTrace(id) {
|
|
88
|
+
const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(id)}`);
|
|
89
|
+
return res.trace;
|
|
90
|
+
}
|
|
91
|
+
/**
|
|
92
|
+
* Delete a conversation.
|
|
93
|
+
*
|
|
94
|
+
* It also leaves every training set built from it, including sets already
|
|
95
|
+
* exported -- so a customer asking you to delete a conversation genuinely
|
|
96
|
+
* removes it from what the model will learn next. Training runs that have
|
|
97
|
+
* already finished are not affected.
|
|
98
|
+
*/
|
|
99
|
+
async deleteTrace(id) {
|
|
100
|
+
return this._http.fetchDelete(`/api/loop/traces/${encodeURIComponent(id)}`);
|
|
101
|
+
}
|
|
102
|
+
// ── feedback ──────────────────────────────────────────────────────────
|
|
103
|
+
/**
|
|
104
|
+
* Record a verdict on a conversation.
|
|
105
|
+
*
|
|
106
|
+
* Append-only: posting a second verdict does not replace the first. Two
|
|
107
|
+
* reviewers disagreeing about an answer is information worth keeping.
|
|
108
|
+
*
|
|
109
|
+
* `source` says who judged: `human`, `verifier` (a deterministic check --
|
|
110
|
+
* tests passed, schema valid), `judge` (a model grading a model), or
|
|
111
|
+
* `behavioural` (what the user did next). When several disagree, a human
|
|
112
|
+
* outranks a verifier, a verifier outranks a judge, and a judge outranks a
|
|
113
|
+
* behavioural hint.
|
|
114
|
+
*/
|
|
115
|
+
async signal(traceId, params) {
|
|
116
|
+
const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`, params);
|
|
117
|
+
return res.signal;
|
|
118
|
+
}
|
|
119
|
+
/** Every verdict on one conversation, oldest first. */
|
|
120
|
+
async listSignals(traceId) {
|
|
121
|
+
const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`);
|
|
122
|
+
return res.signals;
|
|
123
|
+
}
|
|
124
|
+
// ── alternative answers ───────────────────────────────────────────────
|
|
125
|
+
/**
|
|
126
|
+
* Submit an alternative answer to a prompt already captured.
|
|
127
|
+
*
|
|
128
|
+
* This is what makes preference learning scale. A human rewrite is the best
|
|
129
|
+
* signal there is and the least available -- somebody has to sit down and
|
|
130
|
+
* write it. Sample the model several times for the same prompt, score the
|
|
131
|
+
* samples, and a DPO pair falls out automatically: best becomes chosen,
|
|
132
|
+
* worst becomes rejected. Point a stronger model at the prompt instead and
|
|
133
|
+
* the same machinery does distillation.
|
|
134
|
+
*
|
|
135
|
+
* Score them, or they cannot pair: an unscored alternative says nothing about
|
|
136
|
+
* which answer anybody prefers. `score_source` is required alongside a score,
|
|
137
|
+
* because precedence between a verifier, a judge and a person is the whole
|
|
138
|
+
* reason we record who judged.
|
|
139
|
+
*
|
|
140
|
+
* A human correction always outranks any score.
|
|
141
|
+
*
|
|
142
|
+
* @example
|
|
143
|
+
* ```ts
|
|
144
|
+
* for (const sample of await sampleMyModel(prompt, 4)) {
|
|
145
|
+
* await client.loop.addCandidate(traceId, {
|
|
146
|
+
* completion: sample.text,
|
|
147
|
+
* score: await myVerifier(sample.text),
|
|
148
|
+
* score_source: 'verifier',
|
|
149
|
+
* });
|
|
150
|
+
* }
|
|
151
|
+
* ```
|
|
152
|
+
*/
|
|
153
|
+
async addCandidate(traceId, params) {
|
|
154
|
+
const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/candidates`, params);
|
|
155
|
+
return res.candidate;
|
|
156
|
+
}
|
|
157
|
+
/** Every alternative answer recorded for one prompt, oldest first. */
|
|
158
|
+
async listCandidates(traceId) {
|
|
159
|
+
const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/candidates`);
|
|
160
|
+
return res.candidates;
|
|
161
|
+
}
|
|
162
|
+
// ── training sets ─────────────────────────────────────────────────────
|
|
163
|
+
/**
|
|
164
|
+
* Build a training set from the feedback recorded so far.
|
|
165
|
+
*
|
|
166
|
+
* The three methods need genuinely different things, so one set cannot be
|
|
167
|
+
* reshaped into another afterwards:
|
|
168
|
+
*
|
|
169
|
+
* - `sft` -- answers you marked right, and answers you rewrote.
|
|
170
|
+
* - `dpo` -- answers you REWROTE, so a better and a worse version of the same
|
|
171
|
+
* reply exist. Nothing else produces a pair.
|
|
172
|
+
* - `grpo` -- answers with a value or fact they can be checked against.
|
|
173
|
+
*
|
|
174
|
+
* `holdout_percent` holds back your most recent work rather than a random
|
|
175
|
+
* slice, so the evaluation measures whether the model generalised instead of
|
|
176
|
+
* memorised the same week.
|
|
177
|
+
*
|
|
178
|
+
* The result always reports `rejected_counts`: why rows were left out. A
|
|
179
|
+
* small set with a reason is useful; a small set without one is just alarming.
|
|
180
|
+
*/
|
|
181
|
+
async createDataset(params) {
|
|
182
|
+
const res = await this._http.fetchPost('/api/loop/datasets', params);
|
|
183
|
+
return res.dataset;
|
|
184
|
+
}
|
|
185
|
+
/** List training sets, newest first. */
|
|
186
|
+
async listDatasets(params = {}) {
|
|
187
|
+
const q = new URLSearchParams();
|
|
188
|
+
if (params.method)
|
|
189
|
+
q.set('method', params.method);
|
|
190
|
+
if (params.limit != null)
|
|
191
|
+
q.set('limit', String(params.limit));
|
|
192
|
+
if (params.offset != null)
|
|
193
|
+
q.set('offset', String(params.offset));
|
|
194
|
+
const qs = q.toString();
|
|
195
|
+
const res = await this._http.fetchGet(`/api/loop/datasets${qs ? `?${qs}` : ''}`);
|
|
196
|
+
return res.datasets;
|
|
197
|
+
}
|
|
198
|
+
/** Read one training set and its curation report. */
|
|
199
|
+
async getDataset(id) {
|
|
200
|
+
const res = await this._http.fetchGet(`/api/loop/datasets/${encodeURIComponent(id)}`);
|
|
201
|
+
return res.dataset;
|
|
202
|
+
}
|
|
203
|
+
/**
|
|
204
|
+
* Page through the rows a training set actually contains.
|
|
205
|
+
*
|
|
206
|
+
* Worth reading before you spend money on a run: each row carries the address
|
|
207
|
+
* of the conversation it was built from.
|
|
208
|
+
*/
|
|
209
|
+
async listDatasetItems(id, params = {}) {
|
|
210
|
+
const q = new URLSearchParams();
|
|
211
|
+
if (params.limit != null)
|
|
212
|
+
q.set('limit', String(params.limit));
|
|
213
|
+
if (params.offset != null)
|
|
214
|
+
q.set('offset', String(params.offset));
|
|
215
|
+
const qs = q.toString();
|
|
216
|
+
const res = await this._http.fetchGet(`/api/loop/datasets/${encodeURIComponent(id)}/items${qs ? `?${qs}` : ''}`);
|
|
217
|
+
return res.items;
|
|
218
|
+
}
|
|
219
|
+
/**
|
|
220
|
+
* Download a training set as JSONL -- one training row per line, the format
|
|
221
|
+
* every trainer in this space reads.
|
|
222
|
+
*
|
|
223
|
+
* `split` defaults to the training rows; pass `holdout` for the slice held
|
|
224
|
+
* back, or `all` for both.
|
|
225
|
+
*/
|
|
226
|
+
async downloadDataset(id, split) {
|
|
227
|
+
const qs = split ? `?split=${split}` : '';
|
|
228
|
+
return this._http.fetchGet(`/api/loop/datasets/${encodeURIComponent(id)}/download${qs}`);
|
|
229
|
+
}
|
|
230
|
+
/**
|
|
231
|
+
* Delete a training set.
|
|
232
|
+
*
|
|
233
|
+
* The conversations it was built from are untouched -- a set is a selection,
|
|
234
|
+
* and discarding the selection must not discard the evidence.
|
|
235
|
+
*/
|
|
236
|
+
async deleteDataset(id) {
|
|
237
|
+
return this._http.fetchDelete(`/api/loop/datasets/${encodeURIComponent(id)}`);
|
|
238
|
+
}
|
|
239
|
+
// ── capture settings ──────────────────────────────────────────────────
|
|
240
|
+
/** Every source this workspace has configured. */
|
|
241
|
+
async listConfigs() {
|
|
242
|
+
const res = await this._http.fetchGet('/api/loop/configs');
|
|
243
|
+
return res.configs;
|
|
244
|
+
}
|
|
245
|
+
/**
|
|
246
|
+
* Read the capture setting for one source. A source nobody has configured
|
|
247
|
+
* reads back as disabled rather than missing, because "we are not recording
|
|
248
|
+
* this" is the honest answer.
|
|
249
|
+
*/
|
|
250
|
+
async getConfig(source) {
|
|
251
|
+
const res = await this._http.fetchGet(`/api/loop/configs/${encodeURIComponent(source)}`);
|
|
252
|
+
return res.config;
|
|
253
|
+
}
|
|
254
|
+
/**
|
|
255
|
+
* Turn capture on or off for a source, and choose how long it is kept.
|
|
256
|
+
*
|
|
257
|
+
* This is the consent decision the whole feature rests on: nothing is
|
|
258
|
+
* recorded until it is made, and the row remembers who made it and when.
|
|
259
|
+
* `sample_rate` below 1 records a deterministic fraction, chosen so that every
|
|
260
|
+
* turn of one conversation is captured or skipped together -- half a
|
|
261
|
+
* conversation makes training rows with holes in them.
|
|
262
|
+
*/
|
|
263
|
+
async setConfig(source, params) {
|
|
264
|
+
const res = await this._http.fetchPut(`/api/loop/configs/${encodeURIComponent(source)}`, params);
|
|
265
|
+
return res.config;
|
|
266
|
+
}
|
|
267
|
+
/**
|
|
268
|
+
* How much there is, and how much of it each method could actually use.
|
|
269
|
+
*
|
|
270
|
+
* The `ready` numbers are upper bounds: duplicate questions are folded into
|
|
271
|
+
* one row while a set is built, so the finished count can be lower.
|
|
272
|
+
*/
|
|
273
|
+
async stats() {
|
|
274
|
+
const res = await this._http.fetchGet('/api/loop/stats');
|
|
275
|
+
return res.stats;
|
|
276
|
+
}
|
|
277
|
+
}
|
package/dist/types.d.ts
CHANGED
|
@@ -1819,7 +1819,7 @@ export interface ApiKey {
|
|
|
1819
1819
|
* on purpose: a scope added to the catalog must not make an existing SDK build
|
|
1820
1820
|
* reject a key it just read back from the API.
|
|
1821
1821
|
*/
|
|
1822
|
-
export type ApiKeyScope = 'models:read' | 'datasets:read' | 'datasets:write' | 'training:read' | 'training:write' | 'billing:read' | 'deployments:read' | 'deployments:write' | 'analytics:read' | 'integrations:read' | 'integrations:write' | 'skills:read' | 'skills:write' | 'serverless' | (string & {});
|
|
1822
|
+
export type ApiKeyScope = 'models:read' | 'datasets:read' | 'datasets:write' | 'training:read' | 'training:write' | 'billing:read' | 'deployments:read' | 'deployments:write' | 'analytics:read' | 'integrations:read' | 'integrations:write' | 'skills:read' | 'skills:write' | 'loop:read' | 'loop:write' | 'serverless' | (string & {});
|
|
1823
1823
|
/** Result of introspecting an API key — shows what it can do. */
|
|
1824
1824
|
export interface ApiKeyIntrospection {
|
|
1825
1825
|
auth_type: 'api_key' | 'jwt';
|
|
@@ -1955,3 +1955,215 @@ export interface ChatCompletionUsage {
|
|
|
1955
1955
|
total_tokens: number;
|
|
1956
1956
|
prompt_tokens_details?: PromptTokensDetails;
|
|
1957
1957
|
}
|
|
1958
|
+
/** What a training set is shaped for. The three need different things. */
|
|
1959
|
+
export type LoopMethod = 'sft' | 'dpo' | 'grpo';
|
|
1960
|
+
/** Who judged an answer. */
|
|
1961
|
+
export type LoopSource = 'human' | 'verifier' | 'judge' | 'behavioural';
|
|
1962
|
+
/** The judgement itself. */
|
|
1963
|
+
export type LoopVerdict = 'accepted' | 'rejected' | 'edited' | 'scored';
|
|
1964
|
+
/** A function an assistant turn asked to invoke. `arguments` is JSON-encoded. */
|
|
1965
|
+
export interface LoopToolCall {
|
|
1966
|
+
id?: string;
|
|
1967
|
+
type?: string;
|
|
1968
|
+
function: {
|
|
1969
|
+
name: string;
|
|
1970
|
+
arguments: string;
|
|
1971
|
+
};
|
|
1972
|
+
}
|
|
1973
|
+
/** One conversational turn, in the chat-completions shape. */
|
|
1974
|
+
export interface LoopMessage {
|
|
1975
|
+
role: string;
|
|
1976
|
+
content?: string;
|
|
1977
|
+
tool_calls?: LoopToolCall[];
|
|
1978
|
+
tool_call_id?: string;
|
|
1979
|
+
name?: string;
|
|
1980
|
+
}
|
|
1981
|
+
export interface LoopCaptureParams {
|
|
1982
|
+
/** The source being recorded. Must be enabled first, or nothing is stored. */
|
|
1983
|
+
deployment_id: string;
|
|
1984
|
+
model: string;
|
|
1985
|
+
messages: LoopMessage[];
|
|
1986
|
+
completion?: string;
|
|
1987
|
+
/** What the answering turn invoked, if anything. */
|
|
1988
|
+
tool_calls?: LoopToolCall[];
|
|
1989
|
+
/** The tool schema the model was offered. Without it, tool training invents names. */
|
|
1990
|
+
tools?: unknown;
|
|
1991
|
+
model_version?: string;
|
|
1992
|
+
/** Ties multi-turn work together. */
|
|
1993
|
+
conversation_id?: string;
|
|
1994
|
+
/** Your own idempotency handle. Replaying it returns the same trace. */
|
|
1995
|
+
request_id?: string;
|
|
1996
|
+
prompt_tokens?: number;
|
|
1997
|
+
completion_tokens?: number;
|
|
1998
|
+
latency_ms?: number;
|
|
1999
|
+
metadata?: Record<string, unknown>;
|
|
2000
|
+
}
|
|
2001
|
+
export interface LoopCaptureResult {
|
|
2002
|
+
captured: boolean;
|
|
2003
|
+
/** Present when captured. Ours, never the id you sent. */
|
|
2004
|
+
trace_id?: string;
|
|
2005
|
+
/** Set when something was removed before the record was written. */
|
|
2006
|
+
redacted?: boolean;
|
|
2007
|
+
download_url?: string;
|
|
2008
|
+
/** Why nothing was stored, e.g. `capture_disabled`, `not_sampled`. */
|
|
2009
|
+
reason?: string;
|
|
2010
|
+
}
|
|
2011
|
+
export interface LoopSignal {
|
|
2012
|
+
id: string;
|
|
2013
|
+
trace_id: string;
|
|
2014
|
+
source: LoopSource;
|
|
2015
|
+
verdict: LoopVerdict;
|
|
2016
|
+
correction: string | null;
|
|
2017
|
+
score: number | null;
|
|
2018
|
+
ground_truth: string | null;
|
|
2019
|
+
reason: string | null;
|
|
2020
|
+
author: string | null;
|
|
2021
|
+
created_at: string;
|
|
2022
|
+
}
|
|
2023
|
+
export interface LoopSignalParams {
|
|
2024
|
+
verdict: LoopVerdict;
|
|
2025
|
+
source?: LoopSource;
|
|
2026
|
+
/** Required when verdict is `edited`. The answer the model should have given. */
|
|
2027
|
+
correction?: string;
|
|
2028
|
+
/** Required when verdict is `scored`. */
|
|
2029
|
+
score?: number;
|
|
2030
|
+
/** A value or fact the answer can be checked against. GRPO needs one. */
|
|
2031
|
+
ground_truth?: string;
|
|
2032
|
+
reason?: string;
|
|
2033
|
+
author?: string;
|
|
2034
|
+
metadata?: Record<string, unknown>;
|
|
2035
|
+
}
|
|
2036
|
+
export interface LoopTrace {
|
|
2037
|
+
id: string;
|
|
2038
|
+
workspace_id: string;
|
|
2039
|
+
deployment_id: string | null;
|
|
2040
|
+
model: string;
|
|
2041
|
+
model_version: string | null;
|
|
2042
|
+
messages: LoopMessage[];
|
|
2043
|
+
completion: string;
|
|
2044
|
+
tool_calls?: LoopToolCall[];
|
|
2045
|
+
tools?: unknown;
|
|
2046
|
+
conversation_id: string | null;
|
|
2047
|
+
request_id: string | null;
|
|
2048
|
+
prompt_tokens: number | null;
|
|
2049
|
+
completion_tokens: number | null;
|
|
2050
|
+
latency_ms: number | null;
|
|
2051
|
+
redacted_at: string | null;
|
|
2052
|
+
/** What the redaction pass removed, counted by rule. */
|
|
2053
|
+
redaction_report?: Record<string, number>;
|
|
2054
|
+
created_at: string;
|
|
2055
|
+
expires_at: string;
|
|
2056
|
+
signals?: LoopSignal[];
|
|
2057
|
+
}
|
|
2058
|
+
export interface LoopTraceListParams {
|
|
2059
|
+
deployment_id?: string;
|
|
2060
|
+
conversation_id?: string;
|
|
2061
|
+
from?: string;
|
|
2062
|
+
to?: string;
|
|
2063
|
+
/** Only conversations that already carry a verdict. */
|
|
2064
|
+
signalled?: boolean;
|
|
2065
|
+
limit?: number;
|
|
2066
|
+
offset?: number;
|
|
2067
|
+
}
|
|
2068
|
+
export interface LoopTraceListResponse {
|
|
2069
|
+
traces: LoopTrace[];
|
|
2070
|
+
total: number;
|
|
2071
|
+
limit: number;
|
|
2072
|
+
offset: number;
|
|
2073
|
+
}
|
|
2074
|
+
export interface LoopDatasetCreateParams {
|
|
2075
|
+
name: string;
|
|
2076
|
+
method: LoopMethod;
|
|
2077
|
+
deployment_id?: string;
|
|
2078
|
+
from?: string;
|
|
2079
|
+
to?: string;
|
|
2080
|
+
/** Holds back your most recent work, not a random slice. 0-50. */
|
|
2081
|
+
holdout_percent?: number;
|
|
2082
|
+
max_items?: number;
|
|
2083
|
+
}
|
|
2084
|
+
export interface LoopDataset {
|
|
2085
|
+
id: string;
|
|
2086
|
+
workspace_id: string;
|
|
2087
|
+
name: string;
|
|
2088
|
+
method: LoopMethod;
|
|
2089
|
+
status: string;
|
|
2090
|
+
spec: Record<string, unknown>;
|
|
2091
|
+
item_count: number;
|
|
2092
|
+
considered_count: number;
|
|
2093
|
+
/** Why rows were left out, by reason. */
|
|
2094
|
+
rejected_counts: Record<string, number>;
|
|
2095
|
+
holdout_count: number;
|
|
2096
|
+
holdout_cutoff: string | null;
|
|
2097
|
+
created_by: string | null;
|
|
2098
|
+
created_at: string;
|
|
2099
|
+
completed_at: string | null;
|
|
2100
|
+
download_url: string;
|
|
2101
|
+
}
|
|
2102
|
+
export interface LoopDatasetListParams {
|
|
2103
|
+
method?: LoopMethod;
|
|
2104
|
+
limit?: number;
|
|
2105
|
+
offset?: number;
|
|
2106
|
+
}
|
|
2107
|
+
export interface LoopDatasetItem {
|
|
2108
|
+
id: number;
|
|
2109
|
+
trace_id: string;
|
|
2110
|
+
split: 'train' | 'holdout';
|
|
2111
|
+
payload: Record<string, unknown>;
|
|
2112
|
+
dedup_key: string;
|
|
2113
|
+
created_at: string;
|
|
2114
|
+
/** The conversation this row was built from. */
|
|
2115
|
+
trace_url: string;
|
|
2116
|
+
}
|
|
2117
|
+
export interface LoopConfig {
|
|
2118
|
+
deployment_id: string;
|
|
2119
|
+
workspace_id: string;
|
|
2120
|
+
enabled: boolean;
|
|
2121
|
+
retention_days: number;
|
|
2122
|
+
sample_rate: number;
|
|
2123
|
+
enabled_by: string | null;
|
|
2124
|
+
enabled_at: string | null;
|
|
2125
|
+
updated_at: string;
|
|
2126
|
+
}
|
|
2127
|
+
export interface LoopConfigParams {
|
|
2128
|
+
enabled: boolean;
|
|
2129
|
+
retention_days?: number;
|
|
2130
|
+
/** Deterministic per conversation, so turns are never split. 0 < rate <= 1. */
|
|
2131
|
+
sample_rate?: number;
|
|
2132
|
+
}
|
|
2133
|
+
export interface LoopStats {
|
|
2134
|
+
traces: number;
|
|
2135
|
+
signalled_traces: number;
|
|
2136
|
+
signals: number;
|
|
2137
|
+
by_verdict: Record<string, number>;
|
|
2138
|
+
by_source: Record<string, number>;
|
|
2139
|
+
datasets: number;
|
|
2140
|
+
/** Upper bounds: deduplication runs when a set is built. */
|
|
2141
|
+
ready: Record<LoopMethod, number>;
|
|
2142
|
+
}
|
|
2143
|
+
export interface LoopCandidateParams {
|
|
2144
|
+
completion?: string;
|
|
2145
|
+
/** An alternative that is itself a tool call. */
|
|
2146
|
+
tool_calls?: LoopToolCall[];
|
|
2147
|
+
/** Which model produced this alternative. The teacher, when distilling. */
|
|
2148
|
+
model?: string;
|
|
2149
|
+
model_version?: string;
|
|
2150
|
+
/** Unscored alternatives cannot pair -- a missing score is not a low one. */
|
|
2151
|
+
score?: number;
|
|
2152
|
+
/** Required alongside a score: human | verifier | judge | behavioural. */
|
|
2153
|
+
score_source?: LoopSource;
|
|
2154
|
+
reason?: string;
|
|
2155
|
+
metadata?: Record<string, unknown>;
|
|
2156
|
+
}
|
|
2157
|
+
export interface LoopCandidate {
|
|
2158
|
+
id: string;
|
|
2159
|
+
trace_id: string;
|
|
2160
|
+
model: string | null;
|
|
2161
|
+
model_version: string | null;
|
|
2162
|
+
completion: string;
|
|
2163
|
+
tool_calls?: LoopToolCall[];
|
|
2164
|
+
score: number | null;
|
|
2165
|
+
score_source: LoopSource | null;
|
|
2166
|
+
reason: string | null;
|
|
2167
|
+
metadata: Record<string, unknown>;
|
|
2168
|
+
created_at: string;
|
|
2169
|
+
}
|