runbios-sdk 0.2.17-rc.269 → 0.2.18-dev.271
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/client.js +1 -1
- package/dist/index.d.ts +2 -2
- package/dist/index.js +1 -1
- package/dist/resources/datasets.d.ts +1 -25
- package/dist/resources/datasets.js +0 -39
- package/dist/resources/inference.js +1 -1
- package/dist/resources/loop.d.ts +145 -543
- package/dist/resources/loop.js +199 -723
- package/dist/types.d.ts +620 -1070
- package/dist/types.js +1 -1
- package/package.json +1 -1
package/dist/resources/loop.js
CHANGED
|
@@ -1,10 +1,16 @@
|
|
|
1
1
|
/**
|
|
2
|
-
* The Conscious Loop --
|
|
3
|
-
*
|
|
2
|
+
* The Conscious Loop -- pipelines that keep improving a model from the
|
|
3
|
+
* conversations you send them.
|
|
4
4
|
*
|
|
5
|
-
*
|
|
6
|
-
*
|
|
7
|
-
*
|
|
5
|
+
* A pipeline is one use case: a model, the tags whose samples train it, when
|
|
6
|
+
* to train, and how a new version is promoted. Every time it trains it makes
|
|
7
|
+
* an attempt; an attempt that beats the live model becomes the next version.
|
|
8
|
+
*
|
|
9
|
+
* Nothing is captured until a source is switched on, and you can switch it
|
|
10
|
+
* off again at any time. Creating a pipeline switches on the source named by
|
|
11
|
+
* its own tag (`own_tag`), stamping that tag, and deleting it switches that
|
|
12
|
+
* source off again. Anything that looks like a credential or a personal detail
|
|
13
|
+
* is removed before the record is written, never afterwards.
|
|
8
14
|
*
|
|
9
15
|
* Requires an API key carrying `loop:read` and/or `loop:write`. Those scopes are
|
|
10
16
|
* deliberately absent from the Read Only preset, because what they return is
|
|
@@ -17,28 +23,30 @@
|
|
|
17
23
|
*
|
|
18
24
|
* @example
|
|
19
25
|
* ```ts
|
|
20
|
-
* // 1.
|
|
21
|
-
* await client.loop.
|
|
26
|
+
* // 1. A pipeline: a model it can train, and its own tag.
|
|
27
|
+
* const { models } = await client.loop.listModels();
|
|
28
|
+
* const { pipeline } = await client.loop.createPipeline({
|
|
29
|
+
* name: 'Support replies',
|
|
30
|
+
* model: { id: models[0].id },
|
|
31
|
+
* });
|
|
22
32
|
*
|
|
23
|
-
* // 2. Send
|
|
24
|
-
* //
|
|
25
|
-
* const
|
|
26
|
-
* deployment_id:
|
|
33
|
+
* // 2. Send samples from your app to the pipeline's own source: its create
|
|
34
|
+
* // switched that source on, and it tags everything it records.
|
|
35
|
+
* const sent = await client.loop.capture({
|
|
36
|
+
* deployment_id: pipeline.own_tag!,
|
|
27
37
|
* model: 'gpt-4o',
|
|
28
38
|
* messages: [{ role: 'user', content: 'what is our refund window' }],
|
|
29
39
|
* completion: 'thirty days',
|
|
30
40
|
* });
|
|
31
41
|
*
|
|
32
|
-
* // 3. Say whether it was right
|
|
33
|
-
* //
|
|
34
|
-
*
|
|
35
|
-
*
|
|
36
|
-
*
|
|
37
|
-
* });
|
|
42
|
+
* // 3. Say whether it was right, on the trace_id the capture answered with.
|
|
43
|
+
* // A correction is the most valuable answer.
|
|
44
|
+
* if (sent.trace_id) {
|
|
45
|
+
* await client.loop.correct(sent.trace_id, 'Thirty days from delivery, no questions asked.');
|
|
46
|
+
* }
|
|
38
47
|
*
|
|
39
|
-
* // 4.
|
|
40
|
-
* const
|
|
41
|
-
* const jsonl = await client.loop.downloadDataset(ds.id);
|
|
48
|
+
* // 4. Once it has its minimum, it trains; versions appear on the pipeline.
|
|
49
|
+
* const p = await client.loop.getPipeline(pipeline.rule_id);
|
|
42
50
|
* ```
|
|
43
51
|
*/
|
|
44
52
|
export class Loop {
|
|
@@ -58,11 +66,17 @@ export class Loop {
|
|
|
58
66
|
*
|
|
59
67
|
* The returned `trace_id` is OURS. If you pass your own `request_id` it is
|
|
60
68
|
* kept as an idempotency handle: sending the same one again returns the same
|
|
61
|
-
* trace rather than storing a second copy, so a retry is safe.
|
|
69
|
+
* trace rather than storing a second copy, so a retry is safe. Send feedback
|
|
70
|
+
* on that `trace_id`.
|
|
71
|
+
*
|
|
72
|
+
* `deployment_id` is the source. A pipeline's `own_tag` is a source its
|
|
73
|
+
* create switched on, and it stamps that tag on everything it records;
|
|
74
|
+
* `labels` add tags of your own, so the same conversation can feed other
|
|
75
|
+
* pipelines too.
|
|
62
76
|
*
|
|
63
|
-
* When the source is not
|
|
64
|
-
* reason instead of throwing -- capture must never be the thing that
|
|
65
|
-
* your application.
|
|
77
|
+
* When the source is not switched on this resolves with `captured: false`
|
|
78
|
+
* and a reason instead of throwing -- capture must never be the thing that
|
|
79
|
+
* breaks your application.
|
|
66
80
|
*/
|
|
67
81
|
async capture(params) {
|
|
68
82
|
return this._http.fetchPost('/api/loop/traces', params);
|
|
@@ -70,9 +84,9 @@ export class Loop {
|
|
|
70
84
|
/**
|
|
71
85
|
* Bring data you already have into the loop.
|
|
72
86
|
*
|
|
73
|
-
* This does NOT create a training set. It creates conversations,
|
|
74
|
-
* place captured ones live and subject to the same
|
|
75
|
-
*
|
|
87
|
+
* This does NOT create a training set. It creates conversations (samples),
|
|
88
|
+
* in the same place captured ones live and subject to the same feedback and
|
|
89
|
+
* tags. A row becomes trainable when something says it is good, never
|
|
76
90
|
* because it arrived in a file -- which is the one guarantee that separates
|
|
77
91
|
* a corpus from a pile.
|
|
78
92
|
*
|
|
@@ -155,6 +169,12 @@ export class Loop {
|
|
|
155
169
|
q.set('origin', params.origin);
|
|
156
170
|
if (params.unlabelled)
|
|
157
171
|
q.set('unlabelled', 'true');
|
|
172
|
+
if (params.pipeline)
|
|
173
|
+
q.set('pipeline', params.pipeline);
|
|
174
|
+
if (params.signal)
|
|
175
|
+
q.set('signal', params.signal);
|
|
176
|
+
if (params.unsignalled)
|
|
177
|
+
q.set('unsignalled', 'true');
|
|
158
178
|
if (params.limit != null)
|
|
159
179
|
q.set('limit', String(params.limit));
|
|
160
180
|
if (params.offset != null)
|
|
@@ -244,43 +264,17 @@ export class Loop {
|
|
|
244
264
|
const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`);
|
|
245
265
|
return res.signals;
|
|
246
266
|
}
|
|
247
|
-
// ── alternative answers ───────────────────────────────────────────────
|
|
248
267
|
/**
|
|
249
|
-
*
|
|
268
|
+
* The pipelines one conversation feeds: every live pipeline whose tags
|
|
269
|
+
* select it, and whether each is switched on.
|
|
250
270
|
*
|
|
251
|
-
*
|
|
252
|
-
*
|
|
253
|
-
*
|
|
254
|
-
*
|
|
255
|
-
* worst becomes rejected. Point a stronger model at the prompt instead and
|
|
256
|
-
* the same machinery does distillation.
|
|
257
|
-
*
|
|
258
|
-
* Score them, or they cannot pair: an unscored alternative says nothing about
|
|
259
|
-
* which answer anybody prefers. `score_source` is required alongside a score,
|
|
260
|
-
* because precedence between a verifier, a judge and a person is the whole
|
|
261
|
-
* reason we record who judged.
|
|
262
|
-
*
|
|
263
|
-
* A human correction always outranks any score.
|
|
264
|
-
*
|
|
265
|
-
* @example
|
|
266
|
-
* ```ts
|
|
267
|
-
* for (const sample of await sampleMyModel(prompt, 4)) {
|
|
268
|
-
* await client.loop.addCandidate(traceId, {
|
|
269
|
-
* completion: sample.text,
|
|
270
|
-
* score: await myVerifier(sample.text),
|
|
271
|
-
* score_source: 'verifier',
|
|
272
|
-
* });
|
|
273
|
-
* }
|
|
274
|
-
* ```
|
|
271
|
+
* Being listed means the pipeline would select it, not that it will be
|
|
272
|
+
* trained on -- a set still drops answers it cannot use and duplicates, and
|
|
273
|
+
* holds some back to judge by; `note` says so. An empty list is a real
|
|
274
|
+
* answer: the conversation trains nothing until it carries a pipeline's tag.
|
|
275
275
|
*/
|
|
276
|
-
async
|
|
277
|
-
|
|
278
|
-
return res.candidate;
|
|
279
|
-
}
|
|
280
|
-
/** Every alternative answer recorded for one prompt, oldest first. */
|
|
281
|
-
async listCandidates(traceId) {
|
|
282
|
-
const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/candidates`);
|
|
283
|
-
return res.candidates;
|
|
276
|
+
async getTracePipelines(traceId) {
|
|
277
|
+
return this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/pipelines`);
|
|
284
278
|
}
|
|
285
279
|
// ── labels ────────────────────────────────────────────────────────────
|
|
286
280
|
/**
|
|
@@ -344,180 +338,6 @@ export class Loop {
|
|
|
344
338
|
const res = await this._http.fetchGet('/api/loop/labels');
|
|
345
339
|
return res.labels;
|
|
346
340
|
}
|
|
347
|
-
// ── training sets ─────────────────────────────────────────────────────
|
|
348
|
-
/**
|
|
349
|
-
* Build a training set from the feedback recorded so far.
|
|
350
|
-
*
|
|
351
|
-
* The three methods need genuinely different things, so one set cannot be
|
|
352
|
-
* reshaped into another afterwards:
|
|
353
|
-
*
|
|
354
|
-
* - `sft` -- answers you marked right, and answers you rewrote.
|
|
355
|
-
* - `dpo` -- answers you REWROTE, so a better and a worse version of the same
|
|
356
|
-
* reply exist. Nothing else produces a pair.
|
|
357
|
-
* - `grpo` -- answers with a value or fact they can be checked against, or
|
|
358
|
-
* a judge rubric, which scores answers the run has not written yet.
|
|
359
|
-
* - `kto` -- anything carrying a yes or a no, INCLUDING a thumbs-down with
|
|
360
|
-
* nothing written. That row trains nothing under the other three methods,
|
|
361
|
-
* which is why this one exists: it is the feedback people actually give.
|
|
362
|
-
*
|
|
363
|
-
* `holdout_percent` holds back your most recent work rather than a random
|
|
364
|
-
* slice, so the evaluation measures whether the model generalised instead of
|
|
365
|
-
* memorised the same week.
|
|
366
|
-
*
|
|
367
|
-
* The result always reports `rejected_counts`: why rows were left out. A
|
|
368
|
-
* small set with a reason is useful; a small set without one is just alarming.
|
|
369
|
-
*/
|
|
370
|
-
async createDataset(params) {
|
|
371
|
-
const res = await this._http.fetchPost('/api/loop/datasets', params);
|
|
372
|
-
return res.dataset;
|
|
373
|
-
}
|
|
374
|
-
/** List training sets, newest first. */
|
|
375
|
-
// ── build rules ───────────────────────────────────────────────────────
|
|
376
|
-
/**
|
|
377
|
-
* Stand up a rule that builds a set whenever enough new work exists.
|
|
378
|
-
*
|
|
379
|
-
* The manual path asks somebody to notice that enough conversations have
|
|
380
|
-
* been reviewed, remember which filters describe the slice they want, and
|
|
381
|
-
* press build — every time. A rule is that instruction, stored.
|
|
382
|
-
*
|
|
383
|
-
* `spec` is the same selection `createDataset` takes and is replayed
|
|
384
|
-
* verbatim, so an automatic set is identical to a hand-made one.
|
|
385
|
-
*
|
|
386
|
-
* `min_new_rows` counts only work reviewed SINCE THE LAST BUILD. Counting
|
|
387
|
-
* the whole corpus would fire the rule every interval forever, because a
|
|
388
|
-
* total that has crossed a threshold stays across it. It is at least 100,
|
|
389
|
-
* and 100 when absent: a set built from fewer is too small to learn from.
|
|
390
|
-
*/
|
|
391
|
-
async createBuildRule(params) {
|
|
392
|
-
const res = await this._http.fetchPost('/api/loop/build-rules', params);
|
|
393
|
-
return res.rule;
|
|
394
|
-
}
|
|
395
|
-
/**
|
|
396
|
-
* Every build rule, with what each one last did and why.
|
|
397
|
-
*
|
|
398
|
-
* `last_reason` is the field worth reading: a rule quiet because it is
|
|
399
|
-
* waiting looks exactly like one quiet because it is broken.
|
|
400
|
-
*/
|
|
401
|
-
async listBuildRules() {
|
|
402
|
-
const res = await this._http.fetchGet('/api/loop/build-rules');
|
|
403
|
-
return res.rules ?? [];
|
|
404
|
-
}
|
|
405
|
-
/** Stop a standing build rule. Sets it already produced are untouched. */
|
|
406
|
-
async deleteBuildRule(id) {
|
|
407
|
-
return this._http.fetchDelete(`/api/loop/build-rules/${encodeURIComponent(id)}`);
|
|
408
|
-
}
|
|
409
|
-
async listDatasets(params = {}) {
|
|
410
|
-
const q = new URLSearchParams();
|
|
411
|
-
if (params.method)
|
|
412
|
-
q.set('method', params.method);
|
|
413
|
-
if (params.limit != null)
|
|
414
|
-
q.set('limit', String(params.limit));
|
|
415
|
-
if (params.offset != null)
|
|
416
|
-
q.set('offset', String(params.offset));
|
|
417
|
-
const qs = q.toString();
|
|
418
|
-
const res = await this._http.fetchGet(`/api/loop/datasets${qs ? `?${qs}` : ''}`);
|
|
419
|
-
return res.datasets;
|
|
420
|
-
}
|
|
421
|
-
/** Read one training set and its curation report. */
|
|
422
|
-
async getDataset(id) {
|
|
423
|
-
const res = await this._http.fetchGet(`/api/loop/datasets/${encodeURIComponent(id)}`);
|
|
424
|
-
return res.dataset;
|
|
425
|
-
}
|
|
426
|
-
/**
|
|
427
|
-
* Page through the rows a training set actually contains.
|
|
428
|
-
*
|
|
429
|
-
* Worth reading before you spend money on a run: each row carries the address
|
|
430
|
-
* of the conversation it was built from.
|
|
431
|
-
*/
|
|
432
|
-
async listDatasetItems(id, params = {}) {
|
|
433
|
-
const q = new URLSearchParams();
|
|
434
|
-
if (params.limit != null)
|
|
435
|
-
q.set('limit', String(params.limit));
|
|
436
|
-
if (params.offset != null)
|
|
437
|
-
q.set('offset', String(params.offset));
|
|
438
|
-
const qs = q.toString();
|
|
439
|
-
const res = await this._http.fetchGet(`/api/loop/datasets/${encodeURIComponent(id)}/items${qs ? `?${qs}` : ''}`);
|
|
440
|
-
return res.items;
|
|
441
|
-
}
|
|
442
|
-
/**
|
|
443
|
-
* Download a training set as JSONL -- one training row per line, the format
|
|
444
|
-
* every trainer in this space reads.
|
|
445
|
-
*
|
|
446
|
-
* `split` defaults to the training rows; pass `holdout` for the slice held
|
|
447
|
-
* back, or `all` for both.
|
|
448
|
-
*/
|
|
449
|
-
async downloadDataset(id, split) {
|
|
450
|
-
const qs = split ? `?split=${split}` : '';
|
|
451
|
-
return this._http.fetchGet(`/api/loop/datasets/${encodeURIComponent(id)}/download${qs}`);
|
|
452
|
-
}
|
|
453
|
-
/**
|
|
454
|
-
* Delete a training set.
|
|
455
|
-
*
|
|
456
|
-
* The conversations it was built from are untouched -- a set is a selection,
|
|
457
|
-
* and discarding the selection must not discard the evidence. A set a
|
|
458
|
-
* training run was trained on is refused `409 DATASET_IN_USE` and kept, as
|
|
459
|
-
* the record of what that run learned from. See
|
|
460
|
-
* `LoopDatasetDeleteRefusalCode`.
|
|
461
|
-
*/
|
|
462
|
-
async deleteDataset(id) {
|
|
463
|
-
return this._http.fetchDelete(`/api/loop/datasets/${encodeURIComponent(id)}`);
|
|
464
|
-
}
|
|
465
|
-
// ── graders ───────────────────────────────────────────────────────────
|
|
466
|
-
/**
|
|
467
|
-
* Write a rule that scores answers without a person.
|
|
468
|
-
*
|
|
469
|
-
* Human review is the most trustworthy feedback and the least available. A
|
|
470
|
-
* grader is written once and applied to every answer afterwards: did it
|
|
471
|
-
* contain the required phrase, did it parse as the schema you asked for, is
|
|
472
|
-
* the number within tolerance of the known answer, did it call the function
|
|
473
|
-
* it should have.
|
|
474
|
-
*
|
|
475
|
-
* Every kind here is DETERMINISTIC -- no model call, no network. That is why
|
|
476
|
-
* a verifier outranks a judge when they disagree: it cannot be flattered and
|
|
477
|
-
* it cannot drift between runs.
|
|
478
|
-
*
|
|
479
|
-
* `weight` is the multiplier: a criterion that matters twice as much gets
|
|
480
|
-
* twice the weight, and the combined score is the weighted mean over the
|
|
481
|
-
* rules that actually applied.
|
|
482
|
-
*
|
|
483
|
-
* `matches_gold` needs no `expected`: it compares each answer to the gold
|
|
484
|
-
* answer recorded on that same conversation, or the human correction when
|
|
485
|
-
* there is no gold, and scores sampled alternatives against the same gold.
|
|
486
|
-
* A conversation with neither is skipped, not failed, so one rule checks
|
|
487
|
-
* every labelled question without punishing the unlabelled ones. An
|
|
488
|
-
* optional `tolerance` widens the match when both sides are bare numbers.
|
|
489
|
-
*
|
|
490
|
-
* A caution worth knowing before you write a set: a rule made only of
|
|
491
|
-
* `forbidden` phrases is satisfied VACUOUSLY by an answer that says nothing.
|
|
492
|
-
* Pair it with a `required` phrase, or you are rewarding silence.
|
|
493
|
-
*/
|
|
494
|
-
async createGrader(params) {
|
|
495
|
-
const res = await this._http.fetchPost('/api/loop/graders', params);
|
|
496
|
-
return res.grader;
|
|
497
|
-
}
|
|
498
|
-
/** Every rule this workspace has written. */
|
|
499
|
-
async listGraders() {
|
|
500
|
-
const res = await this._http.fetchGet('/api/loop/graders');
|
|
501
|
-
return res.graders;
|
|
502
|
-
}
|
|
503
|
-
async deleteGrader(id) {
|
|
504
|
-
return this._http.fetchDelete(`/api/loop/graders/${encodeURIComponent(id)}`);
|
|
505
|
-
}
|
|
506
|
-
/**
|
|
507
|
-
* Apply this workspace's rules to one captured answer AND to every
|
|
508
|
-
* alternative sampled for it.
|
|
509
|
-
*
|
|
510
|
-
* Grading both in one pass is the point: scoring only the original gives a
|
|
511
|
-
* verdict, while scoring the samples as well gives the preference pair. Sample
|
|
512
|
-
* your model, call this, and you have DPO data with nobody reading anything.
|
|
513
|
-
*
|
|
514
|
-
* A run where no rule could apply writes NOTHING -- no verdict, no scores.
|
|
515
|
-
* Recording a zero that no rule produced would poison curation with a
|
|
516
|
-
* judgement nobody reached, so `applied: 0` is reported instead.
|
|
517
|
-
*/
|
|
518
|
-
async grade(traceId) {
|
|
519
|
-
return this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/grade`);
|
|
520
|
-
}
|
|
521
341
|
// ── capture settings ──────────────────────────────────────────────────
|
|
522
342
|
/** Every source this workspace has configured. */
|
|
523
343
|
async listConfigs() {
|
|
@@ -547,425 +367,40 @@ export class Loop {
|
|
|
547
367
|
return res.config;
|
|
548
368
|
}
|
|
549
369
|
/**
|
|
550
|
-
* How much there is, and how
|
|
370
|
+
* How much there is, and how many samples a pipeline could train on now
|
|
371
|
+
* (`ready.sft`: the ones marked good or corrected).
|
|
551
372
|
*
|
|
552
|
-
*
|
|
553
|
-
*
|
|
373
|
+
* `ready.sft` is an upper bound: duplicates, the held-back set and
|
|
374
|
+
* conversations too long to train on come out of it when an attempt trains.
|
|
554
375
|
*/
|
|
555
376
|
async stats() {
|
|
556
377
|
const res = await this._http.fetchGet('/api/loop/stats');
|
|
557
378
|
return res.stats;
|
|
558
379
|
}
|
|
559
|
-
// ── judges ────────────────────────────────────────────────────────────
|
|
560
|
-
//
|
|
561
|
-
// A grader is a rule: deterministic, cheap, and blind to anything it was not
|
|
562
|
-
// told to look for. A judge is a rubric handed to a model, which is the only
|
|
563
|
-
// thing that can answer "was this answer actually helpful".
|
|
564
|
-
//
|
|
565
|
-
// YOU run the model. This service stores raw prompts and completions and
|
|
566
|
-
// holds them with no outbound credentials at all, which is most of the reason
|
|
567
|
-
// it is safe to store them there -- so it hands the work out instead. Open a
|
|
568
|
-
// run, take the batch, send each `prompt` to whatever model you like, and
|
|
569
|
-
// post the scores back.
|
|
570
|
-
/** Write a rubric: what makes an answer good here, and what to score. */
|
|
571
|
-
async createJudge(params) {
|
|
572
|
-
const res = await this._http.fetchPost('/api/loop/judges', params);
|
|
573
|
-
return res.judge;
|
|
574
|
-
}
|
|
575
|
-
async listJudges() {
|
|
576
|
-
const res = await this._http.fetchGet('/api/loop/judges');
|
|
577
|
-
return res.judges;
|
|
578
|
-
}
|
|
579
|
-
/**
|
|
580
|
-
* Rewrite a rubric. Runs already recorded keep the instructions they used:
|
|
581
|
-
* a verdict has to keep meaning what it meant when it was given.
|
|
582
|
-
*/
|
|
583
|
-
async updateJudge(judgeId, params) {
|
|
584
|
-
const res = await this._http.fetchPut(`/api/loop/judges/${encodeURIComponent(judgeId)}`, params);
|
|
585
|
-
return res.judge;
|
|
586
|
-
}
|
|
587
|
-
/**
|
|
588
|
-
* Retire a rubric. Its runs go with it; the verdicts they wrote stay. A score
|
|
589
|
-
* is evidence about an answer, and it does not stop being true because the
|
|
590
|
-
* rubric was retired.
|
|
591
|
-
*/
|
|
592
|
-
async deleteJudge(judgeId) {
|
|
593
|
-
await this._http.fetchDelete(`/api/loop/judges/${encodeURIComponent(judgeId)}`);
|
|
594
|
-
}
|
|
595
|
-
/**
|
|
596
|
-
* Open a run: select the conversations now, and freeze the rubric onto them.
|
|
597
|
-
*
|
|
598
|
-
* The selection is fixed at this moment on purpose. A run whose selection is
|
|
599
|
-
* a live query silently grows as conversations arrive, so "we scored the
|
|
600
|
-
* refunds slice" becomes a claim about a set that no longer exists.
|
|
601
|
-
*/
|
|
602
|
-
async startRun(judgeId, limit) {
|
|
603
|
-
const res = await this._http.fetchPost(`/api/loop/judges/${encodeURIComponent(judgeId)}/runs`, limit ? { limit } : {});
|
|
604
|
-
return res.run;
|
|
605
|
-
}
|
|
606
|
-
async listRuns(judgeId) {
|
|
607
|
-
const path = judgeId
|
|
608
|
-
? `/api/loop/judges/${encodeURIComponent(judgeId)}/runs`
|
|
609
|
-
: '/api/loop/runs';
|
|
610
|
-
const res = await this._http.fetchGet(path);
|
|
611
|
-
return res.runs;
|
|
612
|
-
}
|
|
613
|
-
async getRun(runId) {
|
|
614
|
-
const res = await this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}`);
|
|
615
|
-
return res.run;
|
|
616
|
-
}
|
|
617
|
-
/**
|
|
618
|
-
* Take the next conversations to score, each rendered into a ready-to-send
|
|
619
|
-
* prompt. Nothing is marked taken, so a caller that dies half way loses
|
|
620
|
-
* nothing: ask again and the same items come back.
|
|
621
|
-
*/
|
|
622
|
-
async takeWork(runId, limit = 20) {
|
|
623
|
-
return this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}/work?limit=${limit}`);
|
|
624
|
-
}
|
|
625
|
-
/**
|
|
626
|
-
* What happened to each conversation in a run, and why.
|
|
627
|
-
*
|
|
628
|
-
* `takeWork` hands out what is still PENDING, so a finished run answers it
|
|
629
|
-
* with an empty list. This answers with every item and the outcome on it:
|
|
630
|
-
* `status`, `scored_at`, and `error` — the reason the caller gave for an
|
|
631
|
-
* item it could not score, which is where a wrong model slug or a refused
|
|
632
|
-
* key actually shows up. A run that ends "scored 0, failed 3" is otherwise a
|
|
633
|
-
* number with no detail behind it.
|
|
634
|
-
*
|
|
635
|
-
* Pass `status: 'failed'` for the usual question. At most 500 items come
|
|
636
|
-
* back at a time; when `has_more` is true, call again with the `next_offset`
|
|
637
|
-
* from the reply.
|
|
638
|
-
*/
|
|
639
|
-
async listRunItems(runId, opts = {}) {
|
|
640
|
-
const qs = new URLSearchParams();
|
|
641
|
-
if (opts.status)
|
|
642
|
-
qs.set('status', opts.status);
|
|
643
|
-
if (opts.limit != null)
|
|
644
|
-
qs.set('limit', String(opts.limit));
|
|
645
|
-
if (opts.offset)
|
|
646
|
-
qs.set('offset', String(opts.offset));
|
|
647
|
-
const suffix = qs.toString() ? `?${qs.toString()}` : '';
|
|
648
|
-
return this._http.fetchGet(`/api/loop/runs/${encodeURIComponent(runId)}/items${suffix}`);
|
|
649
|
-
}
|
|
650
|
-
/**
|
|
651
|
-
* Hand the scores back. A verdict for a conversation outside this run, or
|
|
652
|
-
* scoring something the rubric never asked for, is refused and reported in
|
|
653
|
-
* `rejected` rather than silently dropped.
|
|
654
|
-
*
|
|
655
|
-
* Pass `finish` to close the run in the same call once you have nothing left
|
|
656
|
-
* to send. To close a run WITHOUT scores, call `stopRun`: an empty verdict
|
|
657
|
-
* list with `finish` does the same thing and reads like a mistake.
|
|
658
|
-
*/
|
|
659
|
-
async postVerdicts(runId, verdicts, finish = false) {
|
|
660
|
-
return this._http.fetchPost(`/api/loop/runs/${encodeURIComponent(runId)}/verdicts`, { verdicts, finish });
|
|
661
|
-
}
|
|
662
|
-
/**
|
|
663
|
-
* Close a run that has not finished.
|
|
664
|
-
*
|
|
665
|
-
* A judge may have one open run of its own at a time, so an open run BLOCKS
|
|
666
|
-
* the next one, and the runs that most need closing are the ones nobody can
|
|
667
|
-
* wait out: a run the agent has parked on an empty balance or a refused key
|
|
668
|
-
* stays open until the reason is fixed or somebody stops it.
|
|
669
|
-
*
|
|
670
|
-
* Nothing is deleted. Every verdict already recorded stays recorded, the
|
|
671
|
-
* counters keep saying how much of the selection was covered, and the
|
|
672
|
-
* conversations the run was holding are free for the next one. The run ends
|
|
673
|
-
* as `stopped` rather than `done`, so an interrupted pass and a completed
|
|
674
|
-
* one do not read alike.
|
|
675
|
-
*
|
|
676
|
-
* A run that finished on its own is not rewritten: stopping one answers 409.
|
|
677
|
-
*
|
|
678
|
-
* Returns the stopped run itself, like `startRun` and `getRun`, not the
|
|
679
|
-
* `{run}` envelope the service sends. Every other single-run method in this
|
|
680
|
-
* class unwraps, and `Loop.stop_run` in the Python SDK does too, so a
|
|
681
|
-
* caller who wrote `(await loop.stopRun(id)).status` against one of its
|
|
682
|
-
* siblings is right here as well.
|
|
683
|
-
*/
|
|
684
|
-
async stopRun(runId) {
|
|
685
|
-
const res = await this._http.fetchPost(`/api/loop/runs/${encodeURIComponent(runId)}/stop`, {});
|
|
686
|
-
return res.run;
|
|
687
|
-
}
|
|
688
|
-
// ── the agent ─────────────────────────────────────────────────────────
|
|
689
|
-
//
|
|
690
|
-
// The worker that calls a model on the workspace's behalf: it applies
|
|
691
|
-
// automatic judges and writes sample answers. It spends through a managed
|
|
692
|
-
// serverless key OF THIS WORKSPACE, so every call is billed exactly like
|
|
693
|
-
// one of your own, and turning it off revokes that key.
|
|
694
|
-
/**
|
|
695
|
-
* Can an agent run here, is one running, and is it on for this workspace.
|
|
696
|
-
* `available` is about the environment; `online` about the worker;
|
|
697
|
-
* `credential` about this workspace.
|
|
698
|
-
*/
|
|
699
|
-
async agentStatus() {
|
|
700
|
-
return this._http.fetchGet('/api/loop/agent');
|
|
701
|
-
}
|
|
702
|
-
/**
|
|
703
|
-
* Turn the agent on: mints the workspace's managed serverless key and
|
|
704
|
-
* resumes the automatic judges a previous turn-off paused. After this,
|
|
705
|
-
* automatic judges and sample runs make model calls billed to the
|
|
706
|
-
* workspace. Idempotent.
|
|
707
|
-
*
|
|
708
|
-
* A judge whose model the serving gateway will not route is NOT resumed:
|
|
709
|
-
* making it automatic would buy a run that fails every conversation. It
|
|
710
|
-
* stays paused, `judges_still_paused` counts those, and each one carries
|
|
711
|
-
* `auto_pause_reason` saying so. Point it at a model that is served and
|
|
712
|
-
* turn the agent on again.
|
|
713
|
-
*
|
|
714
|
-
* `monthly_spend_cap_cents` caps what that key may spend on model calls in
|
|
715
|
-
* a calendar month. Omitted = no cap on a fresh key, and an existing cap is
|
|
716
|
-
* left as it is; sent while the agent is already on, it moves the cap on
|
|
717
|
-
* the existing key without minting a new one. When the cap is reached the
|
|
718
|
-
* agent's model calls are refused until next month and its runs pause with
|
|
719
|
-
* that reason.
|
|
720
|
-
*/
|
|
721
|
-
async enableAgent(opts = {}) {
|
|
722
|
-
const body = {};
|
|
723
|
-
if (opts.monthly_spend_cap_cents != null)
|
|
724
|
-
body.monthly_spend_cap_cents = opts.monthly_spend_cap_cents;
|
|
725
|
-
const res = await this._http.fetchPost('/api/loop/agent', body);
|
|
726
|
-
return res.credential;
|
|
727
|
-
}
|
|
728
|
-
/**
|
|
729
|
-
* Turn the agent off: revokes its key and pauses every automatic judge in
|
|
730
|
-
* the workspace (`judges_paused`; each judge shows `auto_paused`). Turning
|
|
731
|
-
* it back on resumes exactly those judges. Open runs stop where they are
|
|
732
|
-
* and continue if it is turned back on. Nothing already scored or written
|
|
733
|
-
* is removed.
|
|
734
|
-
*/
|
|
735
|
-
async disableAgent() {
|
|
736
|
-
return this._http.fetchDelete('/api/loop/agent');
|
|
737
|
-
}
|
|
738
380
|
/**
|
|
739
|
-
*
|
|
740
|
-
*
|
|
741
|
-
*
|
|
742
|
-
*
|
|
743
|
-
*
|
|
744
|
-
*
|
|
745
|
-
* Cost: up to `n` calls to write plus `n` to judge, per conversation, at the
|
|
746
|
-
* workspace's serverless rate; `selection.sample` caps the conversations
|
|
747
|
-
* (max 200) and `n` is capped at 8. Opening a run turns the agent on if it
|
|
748
|
-
* is off.
|
|
381
|
+
* How the loop moved over time: every attempt as a point, oldest first --
|
|
382
|
+
* its attempt number, the version it became (null when it did not), its
|
|
383
|
+
* base model, its win rate, what it cost -- and each day's feedback counted
|
|
384
|
+
* by verdict. `rule_id` narrows the attempts to one pipeline; the feedback
|
|
385
|
+
* counts are the workspace's. `truncated` says older attempts exist beyond
|
|
386
|
+
* `limit`.
|
|
749
387
|
*/
|
|
750
|
-
async
|
|
751
|
-
const res = await this._http.fetchPost('/api/loop/sample-runs', params);
|
|
752
|
-
return res.run;
|
|
753
|
-
}
|
|
754
|
-
async listSampleRuns() {
|
|
755
|
-
const res = await this._http.fetchGet('/api/loop/sample-runs');
|
|
756
|
-
return res.runs;
|
|
757
|
-
}
|
|
758
|
-
async getSampleRun(runId) {
|
|
759
|
-
const res = await this._http.fetchGet(`/api/loop/sample-runs/${encodeURIComponent(runId)}`);
|
|
760
|
-
return res.run;
|
|
761
|
-
}
|
|
762
|
-
// ── automatic training ────────────────────────────────────────────────
|
|
763
|
-
//
|
|
764
|
-
// A training rule is a STANDING INSTRUCTION, not a job. Once it is accepted
|
|
765
|
-
// the platform builds a set, trains a model, books a candidate, compares it
|
|
766
|
-
// against what serves your traffic today and -- if you asked for that --
|
|
767
|
-
// re-points the alias at the winner, on its own schedule, with nobody
|
|
768
|
-
// watching. Every one of those steps spends money from the authorising
|
|
769
|
-
// member's wallet.
|
|
770
|
-
//
|
|
771
|
-
// So the order is always the same: preflight for the estimate and the exact
|
|
772
|
-
// words of the terms, show them to the person whose wallet pays, and only
|
|
773
|
-
// then create with `accept_terms` carrying the `terms_version` they read.
|
|
774
|
-
// Money-bearing edits bump `revision`, clear the consent and stop the rule
|
|
775
|
-
// firing until someone accepts the new amounts.
|
|
776
|
-
//
|
|
777
|
-
// Requires `loop:write` for every mutation, including `preflightTrainingRule`:
|
|
778
|
-
// it is the estimate step of a create, and a key that may not create a rule
|
|
779
|
-
// has no reason to price one.
|
|
780
|
-
/**
|
|
781
|
-
* Price a training rule before anyone agrees to it. Writes nothing.
|
|
782
|
-
*
|
|
783
|
-
* Takes the create body without `accept_terms` and answers with the estimate
|
|
784
|
-
* a member has to see first: the pinned model revision, the worst hourly
|
|
785
|
-
* price each GPU ladder can reach, how many hours each ceiling buys, any
|
|
786
|
-
* refusals that would stop a create, the API keys that cannot follow a
|
|
787
|
-
* cutover because they lack `deployments:read`, and `terms_text` -- the
|
|
788
|
-
* exact sentence to show, with real figures in it.
|
|
789
|
-
*
|
|
790
|
-
* Send the `terms_version` it returns back in `createTrainingRule`. Read the
|
|
791
|
-
* figures out of this response rather than inventing ceilings of your own.
|
|
792
|
-
*
|
|
793
|
-
* `valid: false` with an EMPTY `refusals` list is not nothing: check
|
|
794
|
-
* `unreachable`, which names every peer the platform could not reach. The
|
|
795
|
-
* rule is savable in that state and would be paused, but the estimate around
|
|
796
|
-
* it is not trustworthy -- both `worst_hourly_*` are `0`, and so are both
|
|
797
|
-
* `max_*_hours` beside them -- so do not quote those figures to anyone.
|
|
798
|
-
* `model_revision` is empty unless training-service actually pinned a
|
|
799
|
-
* commit; test `unreachable.length > 0` to tell a degraded pass from a
|
|
800
|
-
* refusal.
|
|
801
|
-
*
|
|
802
|
-
* A `TRAINING_NOT_ENABLED` refusal is neither: training is in beta and this
|
|
803
|
-
* organisation does not hold the grant. It is in `refusals`, never in
|
|
804
|
-
* `unreachable`, and nothing in the request can lift it.
|
|
805
|
-
*/
|
|
806
|
-
async preflightTrainingRule(params) {
|
|
807
|
-
return this._http.fetchPost('/api/loop/training-rules/preflight', params);
|
|
808
|
-
}
|
|
809
|
-
/**
|
|
810
|
-
* Create a training rule and record the consent that pays for it.
|
|
811
|
-
*
|
|
812
|
-
* ACCEPTING THE TERMS AUTHORISES SPENDING FROM YOUR WALLET WHILE YOU ARE NOT
|
|
813
|
-
* PRESENT. From here on the platform may, on its own schedule and without
|
|
814
|
-
* asking again, build a training set, run a training job on rented GPUs,
|
|
815
|
-
* book a second deployment to compare against the one serving your traffic,
|
|
816
|
-
* and pay for the model calls that judge the two. Those charges come out of
|
|
817
|
-
* the wallet of the member whose credentials make this call, up to the
|
|
818
|
-
* ceilings in `money`, and they keep recurring for as long as the rule is
|
|
819
|
-
* enabled. If `promotion.auto_promote` is set, the platform will also
|
|
820
|
-
* re-point your public handle at the new model with nobody reviewing it.
|
|
821
|
-
*
|
|
822
|
-
* `accept_terms` is therefore required, and this method refuses to send the
|
|
823
|
-
* request without it rather than letting the server decide. Call
|
|
824
|
-
* {@link preflightTrainingRule} first, show the person whose wallet pays the
|
|
825
|
-
* `terms_text` and the figures it returns, get an explicit yes, and send the
|
|
826
|
-
* `terms_version` they were shown. Do not invent ceilings or price caps on
|
|
827
|
-
* their behalf.
|
|
828
|
-
*
|
|
829
|
-
* Refused `409 TRAINING_NOT_ENABLED`, with nothing saved, while training is
|
|
830
|
-
* in beta and this organisation does not hold the grant: do not retry, ask
|
|
831
|
-
* the platform team to enable it. See `TrainingRuleSaveRefusalCode`.
|
|
832
|
-
*
|
|
833
|
-
* @example
|
|
834
|
-
* ```ts
|
|
835
|
-
* const estimate = await client.loop.preflightTrainingRule(draft);
|
|
836
|
-
* // show estimate.terms_text and the ceilings to the member, get a yes
|
|
837
|
-
* const rule = await client.loop.createTrainingRule({
|
|
838
|
-
* ...draft,
|
|
839
|
-
* accept_terms: { terms_version: estimate.terms_version },
|
|
840
|
-
* });
|
|
841
|
-
* ```
|
|
842
|
-
*/
|
|
843
|
-
async createTrainingRule(params) {
|
|
844
|
-
// Refused here rather than on the wire: a create without a recorded
|
|
845
|
-
// acceptance is a request to spend somebody's money with no record that
|
|
846
|
-
// they agreed, and the SDK should never be the thing that sends it.
|
|
847
|
-
if (!params.accept_terms || !params.accept_terms.terms_version) {
|
|
848
|
-
throw new Error('RunBiOS: createTrainingRule requires accept_terms.terms_version. '
|
|
849
|
-
+ 'Call preflightTrainingRule, show the member terms_text and the ceilings, '
|
|
850
|
-
+ 'and send back the terms_version they accepted -- this rule spends from their wallet.');
|
|
851
|
-
}
|
|
852
|
-
const res = await this._http.fetchPost('/api/loop/training-rules', params);
|
|
853
|
-
return res.rule;
|
|
854
|
-
}
|
|
855
|
-
/**
|
|
856
|
-
* Every training rule in the workspace, with what each one last did and why.
|
|
857
|
-
*
|
|
858
|
-
* `last_reason` and `paused_reason` are the fields worth reading: a rule
|
|
859
|
-
* quiet because it is waiting looks exactly like one quiet because its
|
|
860
|
-
* consent went stale.
|
|
861
|
-
*/
|
|
862
|
-
async listTrainingRules(params = {}) {
|
|
388
|
+
async getMetricsHistory(params = {}) {
|
|
863
389
|
const q = new URLSearchParams();
|
|
864
|
-
if (params.
|
|
865
|
-
q.set('
|
|
390
|
+
if (params.rule_id)
|
|
391
|
+
q.set('rule_id', params.rule_id);
|
|
866
392
|
if (params.limit != null)
|
|
867
393
|
q.set('limit', String(params.limit));
|
|
868
|
-
if (params.
|
|
869
|
-
q.set('
|
|
394
|
+
if (params.days != null)
|
|
395
|
+
q.set('days', String(params.days));
|
|
870
396
|
const qs = q.toString();
|
|
871
|
-
return this._http.fetchGet(`/api/loop/
|
|
872
|
-
}
|
|
873
|
-
/**
|
|
874
|
-
* Read one rule with its recent runs, what its runs this month have cost at
|
|
875
|
-
* most, and how far its judge agrees with your own reviewers.
|
|
876
|
-
*
|
|
877
|
-
* Returned whole rather than unwrapped to the rule: `month_spent_cents` is
|
|
878
|
-
* the number that says whether the monthly ceiling is about to stop it. It
|
|
879
|
-
* is an upper bound, not an exact spend: see `Pipeline.month_spent_cents`.
|
|
880
|
-
*/
|
|
881
|
-
async getTrainingRule(id) {
|
|
882
|
-
return this._http.fetchGet(`/api/loop/training-rules/${encodeURIComponent(id)}`);
|
|
883
|
-
}
|
|
884
|
-
/**
|
|
885
|
-
* Edit or pause a rule. Absent means unchanged; a null clears the fields
|
|
886
|
-
* that can be cleared.
|
|
887
|
-
*
|
|
888
|
-
* A money-bearing change -- a ceiling, a model, a GPU ladder, the promotion
|
|
889
|
-
* policy, or changing `explore_recipes` in EITHER direction -- bumps
|
|
890
|
-
* `revision`, clears the recorded consent and STOPS the rule firing until
|
|
891
|
-
* someone accepts the new amounts. Turning recipe variants OFF does this
|
|
892
|
-
* too: the pipeline then makes no version at all, variant or not, until the
|
|
893
|
-
* terms are accepted again. The reply says so in
|
|
894
|
-
* `consent_required`, and carries a fresh `preflight` with the new figures.
|
|
895
|
-
* Pass `expected_revision` to be refused with `REVISION_MISMATCH` rather
|
|
896
|
-
* than overwrite an edit somebody else made in the meantime.
|
|
897
|
-
*
|
|
898
|
-
* Switching a rule on, switching it into preference training or carrying
|
|
899
|
-
* `accept_terms` re-runs the preflight, and can be refused
|
|
900
|
-
* `409 TRAINING_NOT_ENABLED` like a create. A rule paused
|
|
901
|
-
* `training_not_enabled` is started again by a save of it once the
|
|
902
|
-
* organisation has been enabled.
|
|
903
|
-
*/
|
|
904
|
-
async updateTrainingRule(id, params) {
|
|
905
|
-
return this._http.fetchPut(`/api/loop/training-rules/${encodeURIComponent(id)}`, params);
|
|
906
|
-
}
|
|
907
|
-
/**
|
|
908
|
-
* Accept the rule's current terms, so it may fire again.
|
|
909
|
-
*
|
|
910
|
-
* ACCEPTING THE TERMS AUTHORISES SPENDING FROM YOUR WALLET WHILE YOU ARE NOT
|
|
911
|
-
* PRESENT, on the amounts as they stand right now. This is the same
|
|
912
|
-
* authorisation {@link createTrainingRule} records, given again because a
|
|
913
|
-
* money-bearing edit cleared the old one: training, a candidate deployment
|
|
914
|
-
* and the judge's model calls are charged to the wallet of the member making
|
|
915
|
-
* this call, up to the rule's ceilings, every time it fires.
|
|
916
|
-
*
|
|
917
|
-
* Show the member the current `terms_text` from a fresh
|
|
918
|
-
* {@link preflightTrainingRule} or from the `preflight` on the update reply,
|
|
919
|
-
* and send the `revision` those figures belong to. A stale revision is
|
|
920
|
-
* refused with `409`, which is the point: it means the amounts moved again
|
|
921
|
-
* after they were read. `409 TRAINING_NOT_ENABLED` means training is in beta
|
|
922
|
-
* and this organisation does not hold the grant.
|
|
923
|
-
*/
|
|
924
|
-
async consentTrainingRule(id, params) {
|
|
925
|
-
const res = await this._http.fetchPost(`/api/loop/training-rules/${encodeURIComponent(id)}/consent`, params);
|
|
926
|
-
return res.rule;
|
|
927
|
-
}
|
|
928
|
-
/**
|
|
929
|
-
* Fire a rule now, without waiting for its cadence.
|
|
930
|
-
*
|
|
931
|
-
* Bypasses the schedule and `min_new_rows` only. The row floors, the money
|
|
932
|
-
* ceilings, the consent, the version limit and the monthly limit all still
|
|
933
|
-
* apply, so this can answer `409 CONSENT_REQUIRED`, `409 RULE_PAUSED`,
|
|
934
|
-
* `409 RUN_ACTIVE`, `409 VERSION_LIMIT_REACHED` (the pipeline has made
|
|
935
|
-
* `max_versions` versions), `409 MONTHLY_LIMIT_REACHED` (the most this
|
|
936
|
-
* month's runs can have cost plus the most one run may cost would pass the
|
|
937
|
-
* monthly limit; the error
|
|
938
|
-
* body carries `month_spent_cents`, `monthly_ceiling_cents`, `run_max_cents`
|
|
939
|
-
* and `resumes_at`), `409 AGENT_OFF` (the workspace's Conscious Loop agent
|
|
940
|
-
* is not turned on, so a trained model could not be compared) or
|
|
941
|
-
* `422 NOT_ENOUGH_ROWS` with the counts it needed. See
|
|
942
|
-
* `TrainingRuleRunRefusalCode`.
|
|
943
|
-
*/
|
|
944
|
-
async runTrainingRule(id) {
|
|
945
|
-
const res = await this._http.fetchPost(`/api/loop/training-rules/${encodeURIComponent(id)}/run`, {});
|
|
946
|
-
return res.run;
|
|
947
|
-
}
|
|
948
|
-
/**
|
|
949
|
-
* Delete a rule. An active run is cancelled; runs that already finished, and
|
|
950
|
-
* anything already promoted, are kept.
|
|
951
|
-
*/
|
|
952
|
-
async deleteTrainingRule(id) {
|
|
953
|
-
return this._http.fetchDelete(`/api/loop/training-rules/${encodeURIComponent(id)}`);
|
|
397
|
+
return this._http.fetchGet(`/api/loop/metrics/history${qs ? `?${qs}` : ''}`);
|
|
954
398
|
}
|
|
955
|
-
|
|
956
|
-
|
|
957
|
-
|
|
958
|
-
|
|
959
|
-
|
|
960
|
-
* build rule is refused with `409`: it would silently retrain the next model
|
|
961
|
-
* on a different source's conversations.
|
|
962
|
-
*/
|
|
963
|
-
async updateBuildRule(id, params) {
|
|
964
|
-
const res = await this._http.fetchPut(`/api/loop/build-rules/${encodeURIComponent(id)}`, params);
|
|
965
|
-
return res.rule;
|
|
966
|
-
}
|
|
967
|
-
// ── training runs ─────────────────────────────────────────────────────
|
|
968
|
-
/** Firings, newest first. Filter by rule or by the state they are sitting in. */
|
|
399
|
+
// ── attempts ──────────────────────────────────────────────────────────
|
|
400
|
+
//
|
|
401
|
+
// Every time a pipeline trains it makes an attempt (a training run). The
|
|
402
|
+
// attempts that beat the live model became its versions.
|
|
403
|
+
/** Attempts, newest first. Filter by pipeline (`rule_id`) or by the state they are sitting in. */
|
|
969
404
|
async listTrainingRuns(params = {}) {
|
|
970
405
|
const q = new URLSearchParams();
|
|
971
406
|
if (params.rule_id)
|
|
@@ -1027,23 +462,22 @@ export class Loop {
|
|
|
1027
462
|
}
|
|
1028
463
|
// ── pipelines ─────────────────────────────────────────────────────────
|
|
1029
464
|
/**
|
|
1030
|
-
* Every
|
|
1031
|
-
*
|
|
1032
|
-
*
|
|
1033
|
-
* flight and which version it will be, and what the pipeline is waiting for.
|
|
1034
|
-
*
|
|
1035
|
-
* Read-only. Every decision stays on the route that owns it -- promote,
|
|
1036
|
-
* reject and roll back on the run, the version limit and `explore_recipes`
|
|
1037
|
-
* on the rule.
|
|
465
|
+
* Every pipeline in the workspace: its versions (the attempts that became
|
|
466
|
+
* the live model), every attempt, which version is live, the attempt in
|
|
467
|
+
* flight, and what the pipeline is waiting for.
|
|
1038
468
|
*
|
|
1039
469
|
* RETURNED WITH ITS ENVELOPE, because the list is not always all of them:
|
|
1040
|
-
* the service returns
|
|
1041
|
-
* workspace and `truncated` is true when more exist
|
|
1042
|
-
* caller that shows `pipelines` alone presents a short list as
|
|
1043
|
-
* it
|
|
470
|
+
* the service returns a page of 100, newest first, `total` is every
|
|
471
|
+
* pipeline in the workspace and `truncated` is true when more exist after
|
|
472
|
+
* this page. A caller that shows `pipelines` alone presents a short list as
|
|
473
|
+
* the whole of it; the next page is `{ offset: offset + pipelines.length }`.
|
|
1044
474
|
*/
|
|
1045
|
-
async listPipelines() {
|
|
1046
|
-
const
|
|
475
|
+
async listPipelines(params = {}) {
|
|
476
|
+
const q = new URLSearchParams();
|
|
477
|
+
if (params.offset)
|
|
478
|
+
q.set('offset', String(params.offset));
|
|
479
|
+
const qs = q.toString();
|
|
480
|
+
const res = await this._http.fetchGet(`/api/loop/pipelines${qs ? `?${qs}` : ''}`);
|
|
1047
481
|
const pipelines = res.pipelines || [];
|
|
1048
482
|
return {
|
|
1049
483
|
pipelines,
|
|
@@ -1053,7 +487,7 @@ export class Loop {
|
|
|
1053
487
|
};
|
|
1054
488
|
}
|
|
1055
489
|
/**
|
|
1056
|
-
* One pipeline, by its
|
|
490
|
+
* One pipeline, by its id.
|
|
1057
491
|
*
|
|
1058
492
|
* `status` says whether it is the platform working (`running`), the member
|
|
1059
493
|
* who has to act (`needs_review`, `needs_funds`), or nothing at all until
|
|
@@ -1066,10 +500,108 @@ export class Loop {
|
|
|
1066
500
|
const res = await this._http.fetchGet(`/api/loop/pipelines/${encodeURIComponent(id)}`);
|
|
1067
501
|
return res.pipeline;
|
|
1068
502
|
}
|
|
503
|
+
// ── creating and changing pipelines ───────────────────────────────────
|
|
504
|
+
//
|
|
505
|
+
// One body -- a name, a model, tags, when to train, how to promote -- and
|
|
506
|
+
// the platform decides the rest: the machines and the limits on each
|
|
507
|
+
// attempt. Saving a pipeline IS the authorization to spend; you pay the GPU
|
|
508
|
+
// time its attempts actually use.
|
|
509
|
+
/**
|
|
510
|
+
* The base models a pipeline can train: only those the engine trains with
|
|
511
|
+
* LoRA SFT and the platform can serve for the comparison, recommended first.
|
|
512
|
+
* The one list {@link createPipeline} accepts a `model.id` from, with the
|
|
513
|
+
* `floors` a new pipeline is held to: the samples its first attempt needs and
|
|
514
|
+
* the smallest `train_when.min_samples`.
|
|
515
|
+
*/
|
|
516
|
+
async listModels() {
|
|
517
|
+
const res = await this._http.fetchGet('/api/loop/models');
|
|
518
|
+
return { ...res, models: res.models || [] };
|
|
519
|
+
}
|
|
520
|
+
/**
|
|
521
|
+
* Create a pipeline -- one use case.
|
|
522
|
+
*
|
|
523
|
+
* SAVING IT AUTHORISES SPENDING FROM YOUR WALLET WHILE YOU ARE NOT PRESENT:
|
|
524
|
+
* whenever its data has grown enough (and at its schedule, when it has one),
|
|
525
|
+
* the platform trains a new version on GPUs it chooses, compares it with what
|
|
526
|
+
* serves your traffic, and promotes it by the pipeline's promotion policy.
|
|
527
|
+
* What is charged is the GPU time the runs actually use; the platform's own
|
|
528
|
+
* per-run and monthly limits stop a run that goes wrong.
|
|
529
|
+
*/
|
|
530
|
+
async createPipeline(params) {
|
|
531
|
+
return this._http.fetchPost('/api/loop/pipelines', params);
|
|
532
|
+
}
|
|
533
|
+
/**
|
|
534
|
+
* Edit a pipeline. Only the fields sent change; `max_versions: null` removes
|
|
535
|
+
* the limit and `challenger_model: null` removes the challenger. `promotion`
|
|
536
|
+
* is how the promotion policy is set. Saving is the authorization, as on a
|
|
537
|
+
* create.
|
|
538
|
+
*/
|
|
539
|
+
async updatePipeline(id, params) {
|
|
540
|
+
return this._http.fetchPut(`/api/loop/pipelines/${encodeURIComponent(id)}`, params);
|
|
541
|
+
}
|
|
542
|
+
/**
|
|
543
|
+
* Delete a pipeline: an attempt still running is cancelled; the versions it
|
|
544
|
+
* made, what serves your app, and its samples and tags are kept. Recording
|
|
545
|
+
* to its own tag stops.
|
|
546
|
+
*/
|
|
547
|
+
async deletePipeline(id) {
|
|
548
|
+
return this._http.fetchDelete(`/api/loop/pipelines/${encodeURIComponent(id)}`);
|
|
549
|
+
}
|
|
550
|
+
/**
|
|
551
|
+
* Train now: skips the schedule, never the minimum. A pipeline short of its
|
|
552
|
+
* samples is refused 422 `NOT_ENOUGH_SAMPLES` with `have` and `need` -- the
|
|
553
|
+
* numbers `minimums` shows. 503 `AGENT_COULD_NOT_START` means the key the
|
|
554
|
+
* attempt would be scored with could not be renewed just then: nothing was
|
|
555
|
+
* started, so press train now again in a minute. Every refusal is a
|
|
556
|
+
* `TrainingRuleRunRefusalCode`.
|
|
557
|
+
*/
|
|
558
|
+
async runPipeline(id) {
|
|
559
|
+
return this._http.fetchPost(`/api/loop/pipelines/${encodeURIComponent(id)}/run`, {});
|
|
560
|
+
}
|
|
561
|
+
/**
|
|
562
|
+
* {@link importRows}, with every row given the pipeline's own tag. `source`
|
|
563
|
+
* defaults to that tag; the rows, the answer and the retry rules are the
|
|
564
|
+
* import door's own. `default_feedback: 'good'` records every answered row
|
|
565
|
+
* that brought no verdict of its own as a good example to learn from.
|
|
566
|
+
*/
|
|
567
|
+
async importIntoPipeline(id, params) {
|
|
568
|
+
return this._http.fetchPost(`/api/loop/pipelines/${encodeURIComponent(id)}/import`, params);
|
|
569
|
+
}
|
|
570
|
+
// ── a pipeline's benchmarks, and its versions side by side ─────────────
|
|
571
|
+
/** The pipeline's benchmarks in priority order (1 is the primary), the ones it stopped using, and its promotion policy. */
|
|
572
|
+
async listPipelineBenchmarks(id) {
|
|
573
|
+
return this._http.fetchGet(`/api/loop/pipelines/${encodeURIComponent(id)}/benchmarks`);
|
|
574
|
+
}
|
|
575
|
+
/** Add a benchmark to a pipeline, last unless `priority` says where. Every version from now on is replayed on it. */
|
|
576
|
+
async attachPipelineBenchmark(id, params) {
|
|
577
|
+
return this._http.fetchPost(`/api/loop/pipelines/${encodeURIComponent(id)}/benchmarks`, params);
|
|
578
|
+
}
|
|
579
|
+
/** Move a benchmark on a pipeline's list; priority 1 makes it the primary. */
|
|
580
|
+
async movePipelineBenchmark(id, benchmarkId, params) {
|
|
581
|
+
return this._http.fetchPut(`/api/loop/pipelines/${encodeURIComponent(id)}/benchmarks/${encodeURIComponent(benchmarkId)}`, params);
|
|
582
|
+
}
|
|
583
|
+
/** Stop using a benchmark. Every score it produced is kept. */
|
|
584
|
+
async removePipelineBenchmark(id, benchmarkId) {
|
|
585
|
+
return this._http.fetchDelete(`/api/loop/pipelines/${encodeURIComponent(id)}/benchmarks/${encodeURIComponent(benchmarkId)}`);
|
|
586
|
+
}
|
|
587
|
+
/**
|
|
588
|
+
* Every version's scores side by side: its head-to-head and every benchmark,
|
|
589
|
+
* `not_measured` where a benchmark was added after it. `a` and `b` are
|
|
590
|
+
* VERSION numbers; with both, `differences` says b minus a per benchmark.
|
|
591
|
+
*/
|
|
592
|
+
async compareVersions(id, params = {}) {
|
|
593
|
+
const q = new URLSearchParams();
|
|
594
|
+
if (params.a != null)
|
|
595
|
+
q.set('a', String(params.a));
|
|
596
|
+
if (params.b != null)
|
|
597
|
+
q.set('b', String(params.b));
|
|
598
|
+
const qs = q.toString();
|
|
599
|
+
return this._http.fetchGet(`/api/loop/pipelines/${encodeURIComponent(id)}/versions/compare${qs ? `?${qs}` : ''}`);
|
|
600
|
+
}
|
|
1069
601
|
// ── the comparison report ─────────────────────────────────────────────
|
|
1070
602
|
/**
|
|
1071
603
|
* The comparison behind a verdict: both models on the same held-out rows,
|
|
1072
|
-
* with identical decoding,
|
|
604
|
+
* with identical decoding, scored by the same judge.
|
|
1073
605
|
*
|
|
1074
606
|
* Read `warnings` before you read `win_rate`. A win rate over a holdout too
|
|
1075
607
|
* small to mean anything, or one measured by a judge that disagrees with
|
|
@@ -1085,8 +617,10 @@ export class Loop {
|
|
|
1085
617
|
* The paired conversations behind the numbers: one prompt, both answers, the
|
|
1086
618
|
* scores each earned, and which won.
|
|
1087
619
|
*
|
|
1088
|
-
* Filter by `winner`
|
|
1089
|
-
*
|
|
620
|
+
* Filter by `winner` -- `candidate` (the attempt won), `serving` (the version
|
|
621
|
+
* serving today won; the item itself says `incumbent`) or `tie` -- and read
|
|
622
|
+
* `serving` first: the pairs the attempt lost are where a verdict is actually
|
|
623
|
+
* checked. `limit` is 50 when absent and at most 200.
|
|
1090
624
|
*/
|
|
1091
625
|
async listEvaluationItems(id, params = {}) {
|
|
1092
626
|
const q = new URLSearchParams();
|
|
@@ -1099,24 +633,6 @@ export class Loop {
|
|
|
1099
633
|
const qs = q.toString();
|
|
1100
634
|
return this._http.fetchGet(`/api/loop/evaluations/${encodeURIComponent(id)}/items${qs ? `?${qs}` : ''}`);
|
|
1101
635
|
}
|
|
1102
|
-
/**
|
|
1103
|
-
* How far a judge agrees with your own reviewers, over the conversations
|
|
1104
|
-
* both have scored.
|
|
1105
|
-
*
|
|
1106
|
-
* This is a gate, not a badge: a rule's `min_judge_agreement` refuses to
|
|
1107
|
-
* promote on the word of a judge that does not agree with the people whose
|
|
1108
|
-
* product it is. `enough_pairs` is the field to read first -- "not enough
|
|
1109
|
-
* reviewer overlap yet" is an answer, and 100% of two pairs is not.
|
|
1110
|
-
*/
|
|
1111
|
-
async getJudgeAgreement(judgeId, params = {}) {
|
|
1112
|
-
const q = new URLSearchParams();
|
|
1113
|
-
if (params.from)
|
|
1114
|
-
q.set('from', params.from);
|
|
1115
|
-
if (params.to)
|
|
1116
|
-
q.set('to', params.to);
|
|
1117
|
-
const qs = q.toString();
|
|
1118
|
-
return this._http.fetchGet(`/api/loop/judges/${encodeURIComponent(judgeId)}/agreement${qs ? `?${qs}` : ''}`);
|
|
1119
|
-
}
|
|
1120
636
|
// ── the standing benchmark ────────────────────────────────────────────
|
|
1121
637
|
/**
|
|
1122
638
|
* Pin a fixed set of conversations, with a judge frozen beside them, and
|
|
@@ -1133,9 +649,10 @@ export class Loop {
|
|
|
1133
649
|
* is refused rather than skipped, because pinning 47 of the 50 you chose is
|
|
1134
650
|
* the set being wrong from the first day and you would never find out.
|
|
1135
651
|
*
|
|
1136
|
-
*
|
|
1137
|
-
*
|
|
1138
|
-
*
|
|
652
|
+
* The platform freezes the judge and the model it scores on (the ones every
|
|
653
|
+
* pipeline comparison uses), gives both models the comparison's own answer
|
|
654
|
+
* length, and sets what one replay may spend, inside the attempt's own
|
|
655
|
+
* scoring limit; none of that is sent.
|
|
1139
656
|
*/
|
|
1140
657
|
async createBenchmark(params) {
|
|
1141
658
|
const res = await this._http.fetchPost('/api/loop/benchmarks', params);
|
|
@@ -1166,7 +683,7 @@ export class Loop {
|
|
|
1166
683
|
return res.benchmark;
|
|
1167
684
|
}
|
|
1168
685
|
/**
|
|
1169
|
-
* The pinned conversations, paged. `limit` is
|
|
686
|
+
* The pinned conversations, paged. `limit` is 50 when absent and at most 200.
|
|
1170
687
|
*
|
|
1171
688
|
* `source_trace_id` and `source_trace_url` come back null once the
|
|
1172
689
|
* conversation a row was copied from has been deleted. The row itself stays
|
|
@@ -1201,7 +718,7 @@ export class Loop {
|
|
|
1201
718
|
return this._http.fetchGet(`/api/loop/benchmarks/${encodeURIComponent(id)}/history${qs ? `?${qs}` : ''}`);
|
|
1202
719
|
}
|
|
1203
720
|
/**
|
|
1204
|
-
* Stop replaying a benchmark, and detach it from every
|
|
721
|
+
* Stop replaying a benchmark, and detach it from every pipeline using it.
|
|
1205
722
|
*
|
|
1206
723
|
* Nothing measured is removed. There is no delete and no update on purpose:
|
|
1207
724
|
* the series of numbers measured against a set is what a benchmark is for,
|
|
@@ -1231,45 +748,4 @@ export class Loop {
|
|
|
1231
748
|
const res = await this._http.fetchGet(`/api/loop/benchmark-runs/${encodeURIComponent(id)}`);
|
|
1232
749
|
return res.benchmark_run;
|
|
1233
750
|
}
|
|
1234
|
-
/**
|
|
1235
|
-
* Attach a benchmark to a rule, or send `{ benchmark_id: null }` to detach
|
|
1236
|
-
* it.
|
|
1237
|
-
*
|
|
1238
|
-
* Its own route rather than a field on the rule body, because it is a
|
|
1239
|
-
* decision to replay a fixed set on every future run of this rule for as
|
|
1240
|
-
* long as it stands, and its refusals -- retired, or belonging to another
|
|
1241
|
-
* workspace -- are about the benchmark rather than about the rule.
|
|
1242
|
-
*
|
|
1243
|
-
* It does not invalidate consent and the reply says so: attaching raises
|
|
1244
|
-
* neither the amount set aside for judge calls nor the amount set aside for
|
|
1245
|
-
* keeping the new model available, so nobody is asked to read the same
|
|
1246
|
-
* sentence again. A retired benchmark is refused with `BENCHMARK_RETIRED`,
|
|
1247
|
-
* because a rule pointed at one would report no number on every run and say
|
|
1248
|
-
* nothing about why.
|
|
1249
|
-
*/
|
|
1250
|
-
async setTrainingRuleBenchmark(id, params) {
|
|
1251
|
-
return this._http.fetchPut(`/api/loop/training-rules/${encodeURIComponent(id)}/benchmark`, params);
|
|
1252
|
-
}
|
|
1253
|
-
// ── agent settings ────────────────────────────────────────────────────
|
|
1254
|
-
/**
|
|
1255
|
-
* The workspace's agent options: which model it defaults to, the system
|
|
1256
|
-
* prompts it judges and samples with, and the monthly cap on what its model
|
|
1257
|
-
* calls may spend.
|
|
1258
|
-
*/
|
|
1259
|
-
async getAgentSettings() {
|
|
1260
|
-
const res = await this._http.fetchGet('/api/loop/agent/settings');
|
|
1261
|
-
return res.settings;
|
|
1262
|
-
}
|
|
1263
|
-
/**
|
|
1264
|
-
* Change them. An absent key leaves that setting exactly where it is; a
|
|
1265
|
-
* present null returns it to the platform default. Those are three
|
|
1266
|
-
* instructions, not two, so `{}` changes nothing.
|
|
1267
|
-
*
|
|
1268
|
-
* `eval_monthly_cap_cents` is pushed to the workspace's managed key, so it
|
|
1269
|
-
* caps what the agent can spend even if a rule's own ceilings are higher.
|
|
1270
|
-
*/
|
|
1271
|
-
async updateAgentSettings(params) {
|
|
1272
|
-
const res = await this._http.fetchPut('/api/loop/agent/settings', params);
|
|
1273
|
-
return res.settings;
|
|
1274
|
-
}
|
|
1275
751
|
}
|