runbios-sdk 0.2.1-dev.97 → 0.2.1-rc.118
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -2
- package/dist/client.js +1 -1
- package/dist/index.d.ts +12 -2
- package/dist/index.js +13 -1
- package/dist/resources/datasets.d.ts +8 -4
- package/dist/resources/datasets.js +15 -4
- package/dist/resources/inference.d.ts +7 -1
- package/dist/resources/inference.js +1 -1
- package/dist/resources/integrations.d.ts +21 -0
- package/dist/resources/integrations.js +20 -0
- package/dist/resources/loop.d.ts +378 -0
- package/dist/resources/loop.js +530 -0
- package/dist/resources/training.d.ts +20 -8
- package/dist/resources/training.js +48 -9
- package/dist/types.d.ts +620 -6
- package/package.json +2 -2
|
@@ -0,0 +1,530 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The Conscious Loop -- capture what your model was asked and answered, record
|
|
3
|
+
* whether it was right, and turn those judgements into training data.
|
|
4
|
+
*
|
|
5
|
+
* Nothing is captured until you turn it on for a source, and you can turn it
|
|
6
|
+
* off again at any time. Anything that looks like a credential or a personal
|
|
7
|
+
* detail is removed before the record is written, never afterwards.
|
|
8
|
+
*
|
|
9
|
+
* Requires an API key carrying `loop:read` and/or `loop:write`. Those scopes are
|
|
10
|
+
* deliberately absent from the Read Only preset, because what they return is
|
|
11
|
+
* your raw prompts and completions rather than catalog metadata.
|
|
12
|
+
*
|
|
13
|
+
* @example
|
|
14
|
+
* ```ts
|
|
15
|
+
* // 1. Decide that this source is recorded.
|
|
16
|
+
* await client.loop.setConfig('my-agent', { enabled: true, retention_days: 30 });
|
|
17
|
+
*
|
|
18
|
+
* // 2. Send what your model was asked and what it answered. Works whoever
|
|
19
|
+
* // served it -- us, another provider, or your own servers.
|
|
20
|
+
* const { trace_id } = await client.loop.capture({
|
|
21
|
+
* deployment_id: 'my-agent',
|
|
22
|
+
* model: 'gpt-4o',
|
|
23
|
+
* messages: [{ role: 'user', content: 'what is our refund window' }],
|
|
24
|
+
* completion: 'thirty days',
|
|
25
|
+
* });
|
|
26
|
+
*
|
|
27
|
+
* // 3. Say whether it was right. A rewrite is the most valuable answer here:
|
|
28
|
+
* // the model learns your version AND learns to avoid its own.
|
|
29
|
+
* await client.loop.signal(trace_id, {
|
|
30
|
+
* verdict: 'edited',
|
|
31
|
+
* correction: 'Thirty days from delivery, no questions asked.',
|
|
32
|
+
* });
|
|
33
|
+
*
|
|
34
|
+
* // 4. Turn the judgements into a training set and take the file.
|
|
35
|
+
* const ds = await client.loop.createDataset({ name: 'support', method: 'sft' });
|
|
36
|
+
* const jsonl = await client.loop.downloadDataset(ds.id);
|
|
37
|
+
* ```
|
|
38
|
+
*/
|
|
39
|
+
export class Loop {
|
|
40
|
+
_http;
|
|
41
|
+
/** @internal */
|
|
42
|
+
constructor(_http) {
|
|
43
|
+
this._http = _http;
|
|
44
|
+
}
|
|
45
|
+
// ── capture ───────────────────────────────────────────────────────────
|
|
46
|
+
/**
|
|
47
|
+
* Send one exchange to be captured.
|
|
48
|
+
*
|
|
49
|
+
* Works regardless of who served the model: ours, OpenAI, Fireworks, your own
|
|
50
|
+
* own servers, an agent framework. Pass `tool_calls` and `tools` when the turn
|
|
51
|
+
* invoked a function -- without them a tool-using exchange trains the model to
|
|
52
|
+
* answer in prose where it should have called something.
|
|
53
|
+
*
|
|
54
|
+
* The returned `trace_id` is OURS. If you pass your own `request_id` it is
|
|
55
|
+
* kept as an idempotency handle: sending the same one again returns the same
|
|
56
|
+
* trace rather than storing a second copy, so a retry is safe.
|
|
57
|
+
*
|
|
58
|
+
* When the source is not enabled this resolves with `captured: false` and a
|
|
59
|
+
* reason instead of throwing -- capture must never be the thing that breaks
|
|
60
|
+
* your application.
|
|
61
|
+
*/
|
|
62
|
+
async capture(params) {
|
|
63
|
+
return this._http.fetchPost('/api/loop/traces', params);
|
|
64
|
+
}
|
|
65
|
+
/**
|
|
66
|
+
* Bring data you already have into the loop.
|
|
67
|
+
*
|
|
68
|
+
* This does NOT create a training set. It creates conversations, in the same
|
|
69
|
+
* place captured ones live and subject to the same review, rules, judges and
|
|
70
|
+
* labels. A row becomes trainable when something says it is good, never
|
|
71
|
+
* because it arrived in a file -- which is the one guarantee that separates
|
|
72
|
+
* a corpus from a pile.
|
|
73
|
+
*
|
|
74
|
+
* Each row is read for what it actually is. A `prompt`/`chosen`/`rejected`
|
|
75
|
+
* triple becomes a preference pair; an answer plus a yes-or-no becomes a
|
|
76
|
+
* thumbs verdict; a question and an answer waits for review; a question with
|
|
77
|
+
* no answer waits for an answer; a paragraph of prose is refused, because it
|
|
78
|
+
* is not a conversation.
|
|
79
|
+
*
|
|
80
|
+
* A verdict that arrives *with* the file is kept -- discarding somebody's
|
|
81
|
+
* judgement would be worse -- but it is recorded as having come from your
|
|
82
|
+
* earlier process rather than from a reviewer here, and the conversation
|
|
83
|
+
* records that it was imported. Neither fact can be reconstructed later, so
|
|
84
|
+
* both are written at the door.
|
|
85
|
+
*
|
|
86
|
+
* `source` is required and becomes the id every imported conversation is
|
|
87
|
+
* filed under: "everything in one bucket called import" is a corpus nobody
|
|
88
|
+
* can slice afterwards. At most 5000 rows per call, because the call is
|
|
89
|
+
* synchronous and somebody is waiting on it.
|
|
90
|
+
*/
|
|
91
|
+
async importRows(params) {
|
|
92
|
+
return this._http.fetchPost('/api/loop/import', params);
|
|
93
|
+
}
|
|
94
|
+
// ── traces ────────────────────────────────────────────────────────────
|
|
95
|
+
/** List captured conversations, newest first. */
|
|
96
|
+
async listTraces(params = {}) {
|
|
97
|
+
const q = new URLSearchParams();
|
|
98
|
+
if (params.deployment_id)
|
|
99
|
+
q.set('deployment_id', params.deployment_id);
|
|
100
|
+
if (params.conversation_id)
|
|
101
|
+
q.set('conversation_id', params.conversation_id);
|
|
102
|
+
if (params.from)
|
|
103
|
+
q.set('from', params.from);
|
|
104
|
+
if (params.to)
|
|
105
|
+
q.set('to', params.to);
|
|
106
|
+
if (params.signalled)
|
|
107
|
+
q.set('signalled', 'true');
|
|
108
|
+
if (params.label)
|
|
109
|
+
q.set('label', params.label);
|
|
110
|
+
// Repeated rather than comma-joined: a value containing a comma would
|
|
111
|
+
// otherwise silently split into two filters that match nothing.
|
|
112
|
+
for (const [k, v] of Object.entries(params.attributes ?? {})) {
|
|
113
|
+
if (k && v)
|
|
114
|
+
q.append('attr', `${k}:${v}`);
|
|
115
|
+
}
|
|
116
|
+
if (params.unlabelled)
|
|
117
|
+
q.set('unlabelled', 'true');
|
|
118
|
+
if (params.limit != null)
|
|
119
|
+
q.set('limit', String(params.limit));
|
|
120
|
+
if (params.offset != null)
|
|
121
|
+
q.set('offset', String(params.offset));
|
|
122
|
+
const qs = q.toString();
|
|
123
|
+
return this._http.fetchGet(`/api/loop/traces${qs ? `?${qs}` : ''}`);
|
|
124
|
+
}
|
|
125
|
+
/** Read one conversation, with every verdict recorded on it. */
|
|
126
|
+
async getTrace(id) {
|
|
127
|
+
const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(id)}`);
|
|
128
|
+
return res.trace;
|
|
129
|
+
}
|
|
130
|
+
/**
|
|
131
|
+
* Delete a conversation.
|
|
132
|
+
*
|
|
133
|
+
* It also leaves every training set built from it, including sets already
|
|
134
|
+
* exported -- so a customer asking you to delete a conversation genuinely
|
|
135
|
+
* removes it from what the model will learn next. Training runs that have
|
|
136
|
+
* already finished are not affected.
|
|
137
|
+
*/
|
|
138
|
+
async deleteTrace(id) {
|
|
139
|
+
return this._http.fetchDelete(`/api/loop/traces/${encodeURIComponent(id)}`);
|
|
140
|
+
}
|
|
141
|
+
// ── feedback ──────────────────────────────────────────────────────────
|
|
142
|
+
/**
|
|
143
|
+
* Record a verdict on a conversation.
|
|
144
|
+
*
|
|
145
|
+
* Append-only: posting a second verdict does not replace the first. Two
|
|
146
|
+
* reviewers disagreeing about an answer is information worth keeping.
|
|
147
|
+
*
|
|
148
|
+
* `source` says who judged: `human`, `verifier` (a deterministic check --
|
|
149
|
+
* tests passed, schema valid), `judge` (a model grading a model), or
|
|
150
|
+
* `behavioural` (what the user did next). When several disagree, a human
|
|
151
|
+
* outranks a verifier, a verifier outranks a judge, and a judge outranks a
|
|
152
|
+
* behavioural hint.
|
|
153
|
+
*/
|
|
154
|
+
async signal(traceId, params) {
|
|
155
|
+
const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`, params);
|
|
156
|
+
return res.signal;
|
|
157
|
+
}
|
|
158
|
+
/**
|
|
159
|
+
* Thumbs up: this answer was good. Shorthand for `signal(id, { verdict:
|
|
160
|
+
* 'accepted' })` — there is one stored vocabulary behind both, so a thumbs-up
|
|
161
|
+
* and an "accepted" can never disagree in your corpus.
|
|
162
|
+
*/
|
|
163
|
+
async upvote(traceId, reason) {
|
|
164
|
+
return this.signal(traceId, { verdict: 'accepted', reason });
|
|
165
|
+
}
|
|
166
|
+
/**
|
|
167
|
+
* Thumbs down: this answer was bad, and you have nothing better to offer.
|
|
168
|
+
*
|
|
169
|
+
* On its own this REMOVES the answer from training rather than teaching
|
|
170
|
+
* anything — there is no better version to learn. If you know what it should
|
|
171
|
+
* have said, `correct()` is worth far more.
|
|
172
|
+
*/
|
|
173
|
+
async downvote(traceId, reason) {
|
|
174
|
+
return this.signal(traceId, { verdict: 'rejected', reason });
|
|
175
|
+
}
|
|
176
|
+
/**
|
|
177
|
+
* The model was wrong; here is what it should have said.
|
|
178
|
+
*
|
|
179
|
+
* The most valuable feedback there is, because it produces BOTH halves of a
|
|
180
|
+
* preference pair from one action: the model learns your answer and learns to
|
|
181
|
+
* avoid its own.
|
|
182
|
+
*/
|
|
183
|
+
async correct(traceId, answer, reason) {
|
|
184
|
+
return this.signal(traceId, { verdict: 'edited', correction: answer, reason });
|
|
185
|
+
}
|
|
186
|
+
/**
|
|
187
|
+
* Record the GOLD answer for this question — the reference, regardless of
|
|
188
|
+
* what the model happened to say.
|
|
189
|
+
*
|
|
190
|
+
* Different from `correct()` in a way that matters: a correction asserts the
|
|
191
|
+
* model was wrong, so the original becomes the rejected half of a pair. A gold
|
|
192
|
+
* answer asserts nothing about the model, so if it MATCHES what was said, no
|
|
193
|
+
* preference pair is invented — but SFT still learns it.
|
|
194
|
+
*
|
|
195
|
+
* It is the most reusable thing you can record: it trains SFT, forms a DPO
|
|
196
|
+
* pair when it differs from the answer, and stands in as the GRPO reference
|
|
197
|
+
* when you have not supplied a separate ground truth.
|
|
198
|
+
*/
|
|
199
|
+
async gold(traceId, answer, reason) {
|
|
200
|
+
return this.signal(traceId, { verdict: 'gold', correction: answer, reason });
|
|
201
|
+
}
|
|
202
|
+
/** Every verdict on one conversation, oldest first. */
|
|
203
|
+
async listSignals(traceId) {
|
|
204
|
+
const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`);
|
|
205
|
+
return res.signals;
|
|
206
|
+
}
|
|
207
|
+
// ── alternative answers ───────────────────────────────────────────────
|
|
208
|
+
/**
|
|
209
|
+
* Submit an alternative answer to a prompt already captured.
|
|
210
|
+
*
|
|
211
|
+
* This is what makes preference learning scale. A human rewrite is the best
|
|
212
|
+
* signal there is and the least available -- somebody has to sit down and
|
|
213
|
+
* write it. Sample the model several times for the same prompt, score the
|
|
214
|
+
* samples, and a DPO pair falls out automatically: best becomes chosen,
|
|
215
|
+
* worst becomes rejected. Point a stronger model at the prompt instead and
|
|
216
|
+
* the same machinery does distillation.
|
|
217
|
+
*
|
|
218
|
+
* Score them, or they cannot pair: an unscored alternative says nothing about
|
|
219
|
+
* which answer anybody prefers. `score_source` is required alongside a score,
|
|
220
|
+
* because precedence between a verifier, a judge and a person is the whole
|
|
221
|
+
* reason we record who judged.
|
|
222
|
+
*
|
|
223
|
+
* A human correction always outranks any score.
|
|
224
|
+
*
|
|
225
|
+
* @example
|
|
226
|
+
* ```ts
|
|
227
|
+
* for (const sample of await sampleMyModel(prompt, 4)) {
|
|
228
|
+
* await client.loop.addCandidate(traceId, {
|
|
229
|
+
* completion: sample.text,
|
|
230
|
+
* score: await myVerifier(sample.text),
|
|
231
|
+
* score_source: 'verifier',
|
|
232
|
+
* });
|
|
233
|
+
* }
|
|
234
|
+
* ```
|
|
235
|
+
*/
|
|
236
|
+
async addCandidate(traceId, params) {
|
|
237
|
+
const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/candidates`, params);
|
|
238
|
+
return res.candidate;
|
|
239
|
+
}
|
|
240
|
+
/** Every alternative answer recorded for one prompt, oldest first. */
|
|
241
|
+
async listCandidates(traceId) {
|
|
242
|
+
const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/candidates`);
|
|
243
|
+
return res.candidates;
|
|
244
|
+
}
|
|
245
|
+
// ── labels ────────────────────────────────────────────────────────────
|
|
246
|
+
/**
|
|
247
|
+
* Label a conversation, so it can be selected later.
|
|
248
|
+
*
|
|
249
|
+
* An average over everything is the least useful thing to train on. A model
|
|
250
|
+
* weak at refunds is fixed with refund examples, and you can only select
|
|
251
|
+
* those if you said so when the conversation arrived.
|
|
252
|
+
*
|
|
253
|
+
* A label may name a `parent`, which is what makes a MASTER GROUP: labelling
|
|
254
|
+
* something `refunds` with parent `billing` makes it selectable as either,
|
|
255
|
+
* and selecting `billing` later gathers every child without you maintaining
|
|
256
|
+
* a list of them.
|
|
257
|
+
*
|
|
258
|
+
* Labels are lowercased and trimmed, so `Refunds` and `refunds` are one label.
|
|
259
|
+
*/
|
|
260
|
+
async label(traceId, labels, parent, attributes) {
|
|
261
|
+
const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/labels`, { labels, parent, attributes });
|
|
262
|
+
return res.labels;
|
|
263
|
+
}
|
|
264
|
+
/**
|
|
265
|
+
* Set named dimensions on a conversation: `{ category: 'billing',
|
|
266
|
+
* language: 'es' }`.
|
|
267
|
+
*
|
|
268
|
+
* An upsert per dimension, so correcting `category` leaves `task` and any
|
|
269
|
+
* bare tags exactly where they were — no delete-then-add.
|
|
270
|
+
*/
|
|
271
|
+
async setAttributes(traceId, attributes) {
|
|
272
|
+
const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/labels`, { attributes });
|
|
273
|
+
return res.labels;
|
|
274
|
+
}
|
|
275
|
+
async unlabel(traceId, label, key) {
|
|
276
|
+
const q = key ? `?key=${encodeURIComponent(key)}` : '';
|
|
277
|
+
return this._http.fetchDelete(`/api/loop/traces/${encodeURIComponent(traceId)}/labels/${encodeURIComponent(label)}${q}`);
|
|
278
|
+
}
|
|
279
|
+
/** Every label in the workspace, with how many conversations carry it. */
|
|
280
|
+
async listLabels() {
|
|
281
|
+
const res = await this._http.fetchGet('/api/loop/labels');
|
|
282
|
+
return res.labels;
|
|
283
|
+
}
|
|
284
|
+
// ── training sets ─────────────────────────────────────────────────────
|
|
285
|
+
/**
|
|
286
|
+
* Build a training set from the feedback recorded so far.
|
|
287
|
+
*
|
|
288
|
+
* The three methods need genuinely different things, so one set cannot be
|
|
289
|
+
* reshaped into another afterwards:
|
|
290
|
+
*
|
|
291
|
+
* - `sft` -- answers you marked right, and answers you rewrote.
|
|
292
|
+
* - `dpo` -- answers you REWROTE, so a better and a worse version of the same
|
|
293
|
+
* reply exist. Nothing else produces a pair.
|
|
294
|
+
* - `grpo` -- answers with a value or fact they can be checked against, or
|
|
295
|
+
* a judge rubric, which scores answers the run has not written yet.
|
|
296
|
+
* - `kto` -- anything carrying a yes or a no, INCLUDING a thumbs-down with
|
|
297
|
+
* nothing written. That row trains nothing under the other three methods,
|
|
298
|
+
* which is why this one exists: it is the feedback people actually give.
|
|
299
|
+
*
|
|
300
|
+
* `holdout_percent` holds back your most recent work rather than a random
|
|
301
|
+
* slice, so the evaluation measures whether the model generalised instead of
|
|
302
|
+
* memorised the same week.
|
|
303
|
+
*
|
|
304
|
+
* The result always reports `rejected_counts`: why rows were left out. A
|
|
305
|
+
* small set with a reason is useful; a small set without one is just alarming.
|
|
306
|
+
*/
|
|
307
|
+
async createDataset(params) {
|
|
308
|
+
const res = await this._http.fetchPost('/api/loop/datasets', params);
|
|
309
|
+
return res.dataset;
|
|
310
|
+
}
|
|
311
|
+
/** List training sets, newest first. */
|
|
312
|
+
async listDatasets(params = {}) {
|
|
313
|
+
const q = new URLSearchParams();
|
|
314
|
+
if (params.method)
|
|
315
|
+
q.set('method', params.method);
|
|
316
|
+
if (params.limit != null)
|
|
317
|
+
q.set('limit', String(params.limit));
|
|
318
|
+
if (params.offset != null)
|
|
319
|
+
q.set('offset', String(params.offset));
|
|
320
|
+
const qs = q.toString();
|
|
321
|
+
const res = await this._http.fetchGet(`/api/loop/datasets${qs ? `?${qs}` : ''}`);
|
|
322
|
+
return res.datasets;
|
|
323
|
+
}
|
|
324
|
+
/** Read one training set and its curation report. */
|
|
325
|
+
async getDataset(id) {
|
|
326
|
+
const res = await this._http.fetchGet(`/api/loop/datasets/${encodeURIComponent(id)}`);
|
|
327
|
+
return res.dataset;
|
|
328
|
+
}
|
|
329
|
+
/**
|
|
330
|
+
* Page through the rows a training set actually contains.
|
|
331
|
+
*
|
|
332
|
+
* Worth reading before you spend money on a run: each row carries the address
|
|
333
|
+
* of the conversation it was built from.
|
|
334
|
+
*/
|
|
335
|
+
async listDatasetItems(id, params = {}) {
|
|
336
|
+
const q = new URLSearchParams();
|
|
337
|
+
if (params.limit != null)
|
|
338
|
+
q.set('limit', String(params.limit));
|
|
339
|
+
if (params.offset != null)
|
|
340
|
+
q.set('offset', String(params.offset));
|
|
341
|
+
const qs = q.toString();
|
|
342
|
+
const res = await this._http.fetchGet(`/api/loop/datasets/${encodeURIComponent(id)}/items${qs ? `?${qs}` : ''}`);
|
|
343
|
+
return res.items;
|
|
344
|
+
}
|
|
345
|
+
/**
|
|
346
|
+
* Download a training set as JSONL -- one training row per line, the format
|
|
347
|
+
* every trainer in this space reads.
|
|
348
|
+
*
|
|
349
|
+
* `split` defaults to the training rows; pass `holdout` for the slice held
|
|
350
|
+
* back, or `all` for both.
|
|
351
|
+
*/
|
|
352
|
+
async downloadDataset(id, split) {
|
|
353
|
+
const qs = split ? `?split=${split}` : '';
|
|
354
|
+
return this._http.fetchGet(`/api/loop/datasets/${encodeURIComponent(id)}/download${qs}`);
|
|
355
|
+
}
|
|
356
|
+
/**
|
|
357
|
+
* Delete a training set.
|
|
358
|
+
*
|
|
359
|
+
* The conversations it was built from are untouched -- a set is a selection,
|
|
360
|
+
* and discarding the selection must not discard the evidence.
|
|
361
|
+
*/
|
|
362
|
+
async deleteDataset(id) {
|
|
363
|
+
return this._http.fetchDelete(`/api/loop/datasets/${encodeURIComponent(id)}`);
|
|
364
|
+
}
|
|
365
|
+
// ── graders ───────────────────────────────────────────────────────────
|
|
366
|
+
/**
|
|
367
|
+
* Write a rule that scores answers without a person.
|
|
368
|
+
*
|
|
369
|
+
* Human review is the most trustworthy feedback and the least available. A
|
|
370
|
+
* grader is written once and applied to every answer afterwards: did it
|
|
371
|
+
* contain the required phrase, did it parse as the schema you asked for, is
|
|
372
|
+
* the number within tolerance of the known answer, did it call the function
|
|
373
|
+
* it should have.
|
|
374
|
+
*
|
|
375
|
+
* Every kind here is DETERMINISTIC -- no model call, no network. That is why
|
|
376
|
+
* a verifier outranks a judge when they disagree: it cannot be flattered and
|
|
377
|
+
* it cannot drift between runs.
|
|
378
|
+
*
|
|
379
|
+
* `weight` is the multiplier: a criterion that matters twice as much gets
|
|
380
|
+
* twice the weight, and the combined score is the weighted mean over the
|
|
381
|
+
* rules that actually applied.
|
|
382
|
+
*
|
|
383
|
+
* A caution worth knowing before you write a set: a rule made only of
|
|
384
|
+
* `forbidden` phrases is satisfied VACUOUSLY by an answer that says nothing.
|
|
385
|
+
* Pair it with a `required` phrase, or you are rewarding silence.
|
|
386
|
+
*/
|
|
387
|
+
async createGrader(params) {
|
|
388
|
+
const res = await this._http.fetchPost('/api/loop/graders', params);
|
|
389
|
+
return res.grader;
|
|
390
|
+
}
|
|
391
|
+
/** Every rule this workspace has written. */
|
|
392
|
+
async listGraders() {
|
|
393
|
+
const res = await this._http.fetchGet('/api/loop/graders');
|
|
394
|
+
return res.graders;
|
|
395
|
+
}
|
|
396
|
+
async deleteGrader(id) {
|
|
397
|
+
return this._http.fetchDelete(`/api/loop/graders/${encodeURIComponent(id)}`);
|
|
398
|
+
}
|
|
399
|
+
/**
|
|
400
|
+
* Apply this workspace's rules to one captured answer AND to every
|
|
401
|
+
* alternative sampled for it.
|
|
402
|
+
*
|
|
403
|
+
* Grading both in one pass is the point: scoring only the original gives a
|
|
404
|
+
* verdict, while scoring the samples as well gives the preference pair. Sample
|
|
405
|
+
* your model, call this, and you have DPO data with nobody reading anything.
|
|
406
|
+
*
|
|
407
|
+
* A run where no rule could apply writes NOTHING -- no verdict, no scores.
|
|
408
|
+
* Recording a zero that no rule produced would poison curation with a
|
|
409
|
+
* judgement nobody reached, so `applied: 0` is reported instead.
|
|
410
|
+
*/
|
|
411
|
+
async grade(traceId) {
|
|
412
|
+
return this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/grade`);
|
|
413
|
+
}
|
|
414
|
+
// ── capture settings ──────────────────────────────────────────────────
|
|
415
|
+
/** Every source this workspace has configured. */
|
|
416
|
+
async listConfigs() {
|
|
417
|
+
const res = await this._http.fetchGet('/api/loop/configs');
|
|
418
|
+
return res.configs;
|
|
419
|
+
}
|
|
420
|
+
/**
|
|
421
|
+
* Read the capture setting for one source. A source nobody has configured
|
|
422
|
+
* reads back as disabled rather than missing, because "we are not recording
|
|
423
|
+
* this" is the honest answer.
|
|
424
|
+
*/
|
|
425
|
+
async getConfig(source) {
|
|
426
|
+
const res = await this._http.fetchGet(`/api/loop/configs/${encodeURIComponent(source)}`);
|
|
427
|
+
return res.config;
|
|
428
|
+
}
|
|
429
|
+
/**
|
|
430
|
+
* Turn capture on or off for a source, and choose how long it is kept.
|
|
431
|
+
*
|
|
432
|
+
* This is the consent decision the whole feature rests on: nothing is
|
|
433
|
+
* recorded until it is made, and the row remembers who made it and when.
|
|
434
|
+
* `sample_rate` below 1 records a deterministic fraction, chosen so that every
|
|
435
|
+
* turn of one conversation is captured or skipped together -- half a
|
|
436
|
+
* conversation makes training rows with holes in them.
|
|
437
|
+
*/
|
|
438
|
+
async setConfig(source, params) {
|
|
439
|
+
const res = await this._http.fetchPut(`/api/loop/configs/${encodeURIComponent(source)}`, params);
|
|
440
|
+
return res.config;
|
|
441
|
+
}
|
|
442
|
+
/**
|
|
443
|
+
* How much there is, and how much of it each method could actually use.
|
|
444
|
+
*
|
|
445
|
+
* The `ready` numbers are upper bounds: duplicate questions are folded into
|
|
446
|
+
* one row while a set is built, so the finished count can be lower.
|
|
447
|
+
*/
|
|
448
|
+
async stats() {
|
|
449
|
+
const res = await this._http.fetchGet('/api/loop/stats');
|
|
450
|
+
return res.stats;
|
|
451
|
+
}
|
|
452
|
+
// ── judges ────────────────────────────────────────────────────────────
|
|
453
|
+
//
|
|
454
|
+
// A grader is a rule: deterministic, cheap, and blind to anything it was not
|
|
455
|
+
// told to look for. A judge is a rubric handed to a model, which is the only
|
|
456
|
+
// thing that can answer "was this answer actually helpful".
|
|
457
|
+
//
|
|
458
|
+
// YOU run the model. This service stores raw prompts and completions and
|
|
459
|
+
// holds them with no outbound credentials at all, which is most of the reason
|
|
460
|
+
// it is safe to store them there -- so it hands the work out instead. Open a
|
|
461
|
+
// run, take the batch, send each `prompt` to whatever model you like, and
|
|
462
|
+
// post the scores back.
|
|
463
|
+
/** Write a rubric: what makes an answer good here, and what to score. */
|
|
464
|
+
async createJudge(params) {
|
|
465
|
+
const res = await this._http.fetchPost('/api/loop/judges', params);
|
|
466
|
+
return res.judge;
|
|
467
|
+
}
|
|
468
|
+
async listJudges() {
|
|
469
|
+
const res = await this._http.fetchGet('/api/loop/judges');
|
|
470
|
+
return res.judges;
|
|
471
|
+
}
|
|
472
|
+
/**
|
|
473
|
+
* Rewrite a rubric. Runs already recorded keep the instructions they used:
|
|
474
|
+
* a verdict has to keep meaning what it meant when it was given.
|
|
475
|
+
*/
|
|
476
|
+
async updateJudge(judgeId, params) {
|
|
477
|
+
const res = await this._http.fetchPut(`/api/loop/judges/${encodeURIComponent(judgeId)}`, params);
|
|
478
|
+
return res.judge;
|
|
479
|
+
}
|
|
480
|
+
/**
|
|
481
|
+
* Retire a rubric. Its runs go with it; the verdicts they wrote stay. A score
|
|
482
|
+
* is evidence about an answer, and it does not stop being true because the
|
|
483
|
+
* rubric was retired.
|
|
484
|
+
*/
|
|
485
|
+
async deleteJudge(judgeId) {
|
|
486
|
+
await this._http.fetchDelete(`/api/loop/judges/${encodeURIComponent(judgeId)}`);
|
|
487
|
+
}
|
|
488
|
+
/**
|
|
489
|
+
* Open a run: select the conversations now, and freeze the rubric onto them.
|
|
490
|
+
*
|
|
491
|
+
* The selection is fixed at this moment on purpose. A run whose selection is
|
|
492
|
+
* a live query silently grows as conversations arrive, so "we scored the
|
|
493
|
+
* refunds slice" becomes a claim about a set that no longer exists.
|
|
494
|
+
*/
|
|
495
|
+
async startRun(judgeId, limit) {
|
|
496
|
+
const res = await this._http.fetchPost(`/api/loop/judges/${encodeURIComponent(judgeId)}/runs`, limit ? { limit } : {});
|
|
497
|
+
return res.run;
|
|
498
|
+
}
|
|
499
|
+
async listRuns(judgeId) {
|
|
500
|
+
const path = judgeId
|
|
501
|
+
? `/api/loop/judges/${encodeURIComponent(judgeId)}/runs`
|
|
502
|
+
: '/api/loop/runs';
|
|
503
|
+
const res = await this._http.fetchGet(path);
|
|
504
|
+
return res.runs;
|
|
505
|
+
}
|
|
506
|
+
async getRun(runId) {
|
|
507
|
+
const res = await this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}`);
|
|
508
|
+
return res.run;
|
|
509
|
+
}
|
|
510
|
+
/**
|
|
511
|
+
* Take the next conversations to score, each rendered into a ready-to-send
|
|
512
|
+
* prompt. Nothing is marked taken, so a caller that dies half way loses
|
|
513
|
+
* nothing: ask again and the same items come back.
|
|
514
|
+
*/
|
|
515
|
+
async takeWork(runId, limit = 20) {
|
|
516
|
+
return this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}/work?limit=${limit}`);
|
|
517
|
+
}
|
|
518
|
+
/**
|
|
519
|
+
* Hand the scores back. A verdict for a conversation outside this run, or
|
|
520
|
+
* scoring something the rubric never asked for, is refused and reported in
|
|
521
|
+
* `rejected` rather than silently dropped.
|
|
522
|
+
*
|
|
523
|
+
* Pass `finish` to close a run you have decided to stop early. Without it an
|
|
524
|
+
* unfinished run stays open, which is the honest state for work that was
|
|
525
|
+
* abandoned rather than completed.
|
|
526
|
+
*/
|
|
527
|
+
async postVerdicts(runId, verdicts, finish = false) {
|
|
528
|
+
return this._http.fetchPost(`/api/loop/runs/${encodeURIComponent(runId)}/verdicts`, { verdicts, finish });
|
|
529
|
+
}
|
|
530
|
+
}
|
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { HttpClient } from '../client.js';
|
|
2
|
-
import type { TrainingCreateParams, TrainingListParams, TrainingJob, TrainingListResponse, TrainingMetrics, TrainingCheckpoint, TrainingLogs, TrainingStopResponse, TrainingResumeResponse, TrainingPreflightResponse, TrainingCapabilities } from '../types.js';
|
|
2
|
+
import type { TrainingCreateParams, TrainingListParams, TrainingJob, TrainingListResponse, TrainingMetrics, TrainingCheckpoint, TrainingLogs, TrainingStopResponse, TrainingResumeResponse, TrainingPreflightResponse, TrainingRecommendParams, TrainingAdvisorVerdict, TrainingCapabilities } from '../types.js';
|
|
3
3
|
/**
|
|
4
4
|
* Create, monitor, and manage fine-tuning training jobs.
|
|
5
5
|
*/
|
|
@@ -12,8 +12,11 @@ export declare class Training {
|
|
|
12
12
|
/**
|
|
13
13
|
* Create a new training job (book-before-reveal).
|
|
14
14
|
*
|
|
15
|
-
*
|
|
16
|
-
*
|
|
15
|
+
* Choose method, adapter, datasets and the complete hyperparameter config
|
|
16
|
+
* first, call {@link preflight}, then choose one returned GPU configuration
|
|
17
|
+
* last. Its minimum is derived from every earlier choice. This call blocks
|
|
18
|
+
* while that placement is booked (~40s typical). A training id and the
|
|
19
|
+
* "training started" email exist only once a real machine is
|
|
17
20
|
* secured, so the returned `status` is one of:
|
|
18
21
|
*
|
|
19
22
|
* - `"booked"` — a GPU was secured (booked == secured); the job then
|
|
@@ -25,11 +28,12 @@ export declare class Training {
|
|
|
25
28
|
* (`queueIfUnavailable: true`); waits for stock at zero charge and books
|
|
26
29
|
* via the same path.
|
|
27
30
|
*
|
|
28
|
-
* If
|
|
29
|
-
*
|
|
30
|
-
*
|
|
31
|
-
*
|
|
32
|
-
*
|
|
31
|
+
* If booking fails, this rejects with `CAPACITY_UNAVAILABLE` (HTTP 409)
|
|
32
|
+
* carrying fresh qualifying alternatives — no phantom job and no charge.
|
|
33
|
+
* Keep every earlier field unchanged, select one alternative and retry. The
|
|
34
|
+
* rejected type is never recommended back. Queue consent is appropriate only
|
|
35
|
+
* when the alternatives list is empty; waiting is unbilled. The SDK never
|
|
36
|
+
* auto-substitutes a GPU; a transient 503 is a retry, never a capacity verdict.
|
|
33
37
|
*
|
|
34
38
|
* @example
|
|
35
39
|
* ```ts
|
|
@@ -51,6 +55,14 @@ export declare class Training {
|
|
|
51
55
|
create(params: TrainingCreateParams): Promise<TrainingJob>;
|
|
52
56
|
/** Validate and canonicalize a training request without creating or billing a job. */
|
|
53
57
|
preflight(params: TrainingCreateParams): Promise<TrainingPreflightResponse>;
|
|
58
|
+
/**
|
|
59
|
+
* Ask the trainer's own sizing model what to run: per-device batch,
|
|
60
|
+
* accumulation, learning rate, warmup, predicted peak memory, minimum GPU
|
|
61
|
+
* count and a wall-clock estimate for this model on this GPU type, with the
|
|
62
|
+
* basis of every number. Side-effect free. When `available` is false no
|
|
63
|
+
* advisor is deployed and the other fields are absent -- nothing is guessed.
|
|
64
|
+
*/
|
|
65
|
+
recommend(params: TrainingRecommendParams): Promise<TrainingAdvisorVerdict>;
|
|
54
66
|
/** Return one server-driven page with pagination metadata. */
|
|
55
67
|
listPage(params?: TrainingListParams): Promise<TrainingListResponse>;
|
|
56
68
|
/**
|
|
@@ -67,8 +67,6 @@ function buildTrainingRequest(params) {
|
|
|
67
67
|
body.integration_id = params.integrationId;
|
|
68
68
|
if (params.networkVolumeId !== undefined)
|
|
69
69
|
body.network_volume_id = params.networkVolumeId;
|
|
70
|
-
if (params.cacheDataset !== undefined)
|
|
71
|
-
body.cache_dataset = params.cacheDataset;
|
|
72
70
|
if (params.datasetSampleLimits !== undefined)
|
|
73
71
|
body.dataset_sample_limits = params.datasetSampleLimits;
|
|
74
72
|
if (params.datasetMixing !== undefined)
|
|
@@ -78,6 +76,8 @@ function buildTrainingRequest(params) {
|
|
|
78
76
|
const config = {};
|
|
79
77
|
if (params.epochs !== undefined)
|
|
80
78
|
config.num_train_epochs = params.epochs;
|
|
79
|
+
if (params.maxSteps !== undefined)
|
|
80
|
+
config.max_steps = params.maxSteps;
|
|
81
81
|
if (params.batchSize !== undefined)
|
|
82
82
|
config.per_device_train_batch_size = params.batchSize;
|
|
83
83
|
if (params.gradientAccumulation !== undefined)
|
|
@@ -183,8 +183,11 @@ export class Training {
|
|
|
183
183
|
/**
|
|
184
184
|
* Create a new training job (book-before-reveal).
|
|
185
185
|
*
|
|
186
|
-
*
|
|
187
|
-
*
|
|
186
|
+
* Choose method, adapter, datasets and the complete hyperparameter config
|
|
187
|
+
* first, call {@link preflight}, then choose one returned GPU configuration
|
|
188
|
+
* last. Its minimum is derived from every earlier choice. This call blocks
|
|
189
|
+
* while that placement is booked (~40s typical). A training id and the
|
|
190
|
+
* "training started" email exist only once a real machine is
|
|
188
191
|
* secured, so the returned `status` is one of:
|
|
189
192
|
*
|
|
190
193
|
* - `"booked"` — a GPU was secured (booked == secured); the job then
|
|
@@ -196,11 +199,12 @@ export class Training {
|
|
|
196
199
|
* (`queueIfUnavailable: true`); waits for stock at zero charge and books
|
|
197
200
|
* via the same path.
|
|
198
201
|
*
|
|
199
|
-
* If
|
|
200
|
-
*
|
|
201
|
-
*
|
|
202
|
-
*
|
|
203
|
-
*
|
|
202
|
+
* If booking fails, this rejects with `CAPACITY_UNAVAILABLE` (HTTP 409)
|
|
203
|
+
* carrying fresh qualifying alternatives — no phantom job and no charge.
|
|
204
|
+
* Keep every earlier field unchanged, select one alternative and retry. The
|
|
205
|
+
* rejected type is never recommended back. Queue consent is appropriate only
|
|
206
|
+
* when the alternatives list is empty; waiting is unbilled. The SDK never
|
|
207
|
+
* auto-substitutes a GPU; a transient 503 is a retry, never a capacity verdict.
|
|
204
208
|
*
|
|
205
209
|
* @example
|
|
206
210
|
* ```ts
|
|
@@ -240,6 +244,41 @@ export class Training {
|
|
|
240
244
|
const { body } = buildTrainingRequest(params);
|
|
241
245
|
return this._http.fetchPost('/api/training/preflight', body);
|
|
242
246
|
}
|
|
247
|
+
/**
|
|
248
|
+
* Ask the trainer's own sizing model what to run: per-device batch,
|
|
249
|
+
* accumulation, learning rate, warmup, predicted peak memory, minimum GPU
|
|
250
|
+
* count and a wall-clock estimate for this model on this GPU type, with the
|
|
251
|
+
* basis of every number. Side-effect free. When `available` is false no
|
|
252
|
+
* advisor is deployed and the other fields are absent -- nothing is guessed.
|
|
253
|
+
*/
|
|
254
|
+
async recommend(params) {
|
|
255
|
+
const q = new URLSearchParams({ model_id: params.model, gpu_type: params.gpuType });
|
|
256
|
+
if (params.modelRevision)
|
|
257
|
+
q.set('model_revision', params.modelRevision);
|
|
258
|
+
if (params.integrationId)
|
|
259
|
+
q.set('integration_id', params.integrationId);
|
|
260
|
+
if (params.gpuCount !== undefined)
|
|
261
|
+
q.set('gpu_count', String(params.gpuCount));
|
|
262
|
+
if (params.adapter)
|
|
263
|
+
q.set('train_type', params.adapter);
|
|
264
|
+
if (params.method)
|
|
265
|
+
q.set('method', params.method === 'cpt' ? 'pt' : params.method);
|
|
266
|
+
if (params.maxLength !== undefined)
|
|
267
|
+
q.set('max_length', String(params.maxLength));
|
|
268
|
+
if (params.epochs !== undefined)
|
|
269
|
+
q.set('num_train_epochs', String(params.epochs));
|
|
270
|
+
if (params.maxSteps !== undefined)
|
|
271
|
+
q.set('max_steps', String(params.maxSteps));
|
|
272
|
+
if (params.perDeviceTrainBatchSize !== undefined)
|
|
273
|
+
q.set('per_device_train_batch_size', String(params.perDeviceTrainBatchSize));
|
|
274
|
+
if (params.gradientAccumulationSteps !== undefined)
|
|
275
|
+
q.set('gradient_accumulation_steps', String(params.gradientAccumulationSteps));
|
|
276
|
+
if (params.datasetIds?.length)
|
|
277
|
+
q.set('dataset_ids', params.datasetIds.join(','));
|
|
278
|
+
if (params.workspaceId)
|
|
279
|
+
q.set('workspace_id', params.workspaceId);
|
|
280
|
+
return this._http.fetchGet(`/api/training/recommend?${q}`);
|
|
281
|
+
}
|
|
243
282
|
/** Return one server-driven page with pagination metadata. */
|
|
244
283
|
async listPage(params = {}) {
|
|
245
284
|
const q = new URLSearchParams();
|