runbios-sdk 0.2.1-rc.98 → 0.2.2-dev.171
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +32 -2
- package/dist/client.d.ts +48 -1
- package/dist/client.js +64 -1
- package/dist/index.d.ts +15 -3
- package/dist/index.js +16 -2
- package/dist/resources/datasets.d.ts +8 -4
- package/dist/resources/datasets.js +15 -4
- package/dist/resources/gpu.js +7 -0
- package/dist/resources/inference.d.ts +33 -2
- package/dist/resources/inference.js +46 -1
- package/dist/resources/integrations.d.ts +21 -0
- package/dist/resources/integrations.js +20 -0
- package/dist/resources/loop.d.ts +850 -0
- package/dist/resources/loop.js +1189 -0
- package/dist/resources/training.d.ts +20 -8
- package/dist/resources/training.js +52 -9
- package/dist/types.d.ts +1980 -6
- package/dist/types.js +11 -1
- package/package.json +2 -2
|
@@ -0,0 +1,850 @@
|
|
|
1
|
+
import type { HttpClient } from '../client.js';
|
|
2
|
+
import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, Benchmark, BenchmarkCreateParams, BenchmarkHistoryParams, BenchmarkHistoryResponse, BenchmarkItemListParams, BenchmarkItemsResponse, BenchmarkListParams, BenchmarkListResponse, BenchmarkRetireParams, BenchmarkRun, TrainingRuleBenchmarkRequest, AgentSettings, AgentSettingsRequest } from '../types.js';
|
|
3
|
+
/**
|
|
4
|
+
* The Conscious Loop -- capture what your model was asked and answered, record
|
|
5
|
+
* whether it was right, and turn those judgements into training data.
|
|
6
|
+
*
|
|
7
|
+
* Nothing is captured until you turn it on for a source, and you can turn it
|
|
8
|
+
* off again at any time. Anything that looks like a credential or a personal
|
|
9
|
+
* detail is removed before the record is written, never afterwards.
|
|
10
|
+
*
|
|
11
|
+
* Requires an API key carrying `loop:read` and/or `loop:write`. Those scopes are
|
|
12
|
+
* deliberately absent from the Read Only preset, because what they return is
|
|
13
|
+
* your raw prompts and completions rather than catalog metadata.
|
|
14
|
+
*
|
|
15
|
+
* @example
|
|
16
|
+
* ```ts
|
|
17
|
+
* // 1. Decide that this source is recorded.
|
|
18
|
+
* await client.loop.setConfig('my-agent', { enabled: true, retention_days: 30 });
|
|
19
|
+
*
|
|
20
|
+
* // 2. Send what your model was asked and what it answered. Works whoever
|
|
21
|
+
* // served it -- us, another provider, or your own servers.
|
|
22
|
+
* const { trace_id } = await client.loop.capture({
|
|
23
|
+
* deployment_id: 'my-agent',
|
|
24
|
+
* model: 'gpt-4o',
|
|
25
|
+
* messages: [{ role: 'user', content: 'what is our refund window' }],
|
|
26
|
+
* completion: 'thirty days',
|
|
27
|
+
* });
|
|
28
|
+
*
|
|
29
|
+
* // 3. Say whether it was right. A rewrite is the most valuable answer here:
|
|
30
|
+
* // the model learns your version AND learns to avoid its own.
|
|
31
|
+
* await client.loop.signal(trace_id, {
|
|
32
|
+
* verdict: 'edited',
|
|
33
|
+
* correction: 'Thirty days from delivery, no questions asked.',
|
|
34
|
+
* });
|
|
35
|
+
*
|
|
36
|
+
* // 4. Turn the judgements into a training set and take the file.
|
|
37
|
+
* const ds = await client.loop.createDataset({ name: 'support', method: 'sft' });
|
|
38
|
+
* const jsonl = await client.loop.downloadDataset(ds.id);
|
|
39
|
+
* ```
|
|
40
|
+
*/
|
|
41
|
+
export declare class Loop {
|
|
42
|
+
private readonly _http;
|
|
43
|
+
/** @internal */
|
|
44
|
+
constructor(_http: HttpClient);
|
|
45
|
+
/**
|
|
46
|
+
* Send one exchange to be captured.
|
|
47
|
+
*
|
|
48
|
+
* Works regardless of who served the model: ours, OpenAI, Fireworks, your own
|
|
49
|
+
* own servers, an agent framework. Pass `tool_calls` and `tools` when the turn
|
|
50
|
+
* invoked a function -- without them a tool-using exchange trains the model to
|
|
51
|
+
* answer in prose where it should have called something.
|
|
52
|
+
*
|
|
53
|
+
* The returned `trace_id` is OURS. If you pass your own `request_id` it is
|
|
54
|
+
* kept as an idempotency handle: sending the same one again returns the same
|
|
55
|
+
* trace rather than storing a second copy, so a retry is safe.
|
|
56
|
+
*
|
|
57
|
+
* When the source is not enabled this resolves with `captured: false` and a
|
|
58
|
+
* reason instead of throwing -- capture must never be the thing that breaks
|
|
59
|
+
* your application.
|
|
60
|
+
*/
|
|
61
|
+
capture(params: LoopCaptureParams): Promise<LoopCaptureResult>;
|
|
62
|
+
/**
|
|
63
|
+
* Bring data you already have into the loop.
|
|
64
|
+
*
|
|
65
|
+
* This does NOT create a training set. It creates conversations, in the same
|
|
66
|
+
* place captured ones live and subject to the same review, rules, judges and
|
|
67
|
+
* labels. A row becomes trainable when something says it is good, never
|
|
68
|
+
* because it arrived in a file -- which is the one guarantee that separates
|
|
69
|
+
* a corpus from a pile.
|
|
70
|
+
*
|
|
71
|
+
* Each row is read for what it actually is. A `prompt`/`chosen`/`rejected`
|
|
72
|
+
* triple becomes a preference pair; an answer plus a yes-or-no becomes a
|
|
73
|
+
* thumbs verdict; a question and an answer waits for review; a question with
|
|
74
|
+
* no answer waits for an answer; a paragraph of prose is refused, because it
|
|
75
|
+
* is not a conversation.
|
|
76
|
+
*
|
|
77
|
+
* A verdict that arrives *with* the file is kept -- discarding somebody's
|
|
78
|
+
* judgement would be worse -- but it is recorded as having come from your
|
|
79
|
+
* earlier process rather than from a reviewer here, and the conversation
|
|
80
|
+
* records that it was imported. Neither fact can be reconstructed later, so
|
|
81
|
+
* both are written at the door.
|
|
82
|
+
*
|
|
83
|
+
* `source` is required and becomes the id every imported conversation is
|
|
84
|
+
* filed under: "everything in one bucket called import" is a corpus nobody
|
|
85
|
+
* can slice afterwards. At most 5000 rows per call, because the call is
|
|
86
|
+
* synchronous and somebody is waiting on it.
|
|
87
|
+
*
|
|
88
|
+
* WHAT COMES BACK. `imported` is what this call created and `trace_ids` names
|
|
89
|
+
* the conversations from the file that are now in the loop, so a caller can
|
|
90
|
+
* label, review or delete them. `by_shape`, `reviewed` and `needs_review`
|
|
91
|
+
* count what THIS CALL created, not what was merely recognised and not rows
|
|
92
|
+
* an earlier import already had, and the sentences in `notes` are written
|
|
93
|
+
* from those counts, so the numbers and the prose describe the same call and
|
|
94
|
+
* a re-send cannot report a reviewed conversation as waiting for review.
|
|
95
|
+
* `refused` is rows whose shape could not be read, which the file fixes;
|
|
96
|
+
* `not_saved` is rows that were understood and then could not be stored,
|
|
97
|
+
* which the file cannot fix.
|
|
98
|
+
*
|
|
99
|
+
* A call with any `not_saved`, or any `verdicts_not_saved` (the conversation
|
|
100
|
+
* stored and the judgement that came with it did not), REJECTS with
|
|
101
|
+
* `LoopImportIncompleteError`, whose `outcome` carries all of the above and
|
|
102
|
+
* whose `notSavedRows` and `verdictsNotSavedRows` name the positions.
|
|
103
|
+
*
|
|
104
|
+
* HOW A RETRY IS SAFE. Every response carries `import_id`, the token this
|
|
105
|
+
* call was filed under. Send the same rows again with it, and rows that
|
|
106
|
+
* already arrived come back on `already_present` instead of being stored
|
|
107
|
+
* twice, while a verdict that failed to write is attempted again. A call that
|
|
108
|
+
* repeats no token is its own import, deliberately: two calls carrying the
|
|
109
|
+
* same rows are as likely to be two pages of one export as one call twice,
|
|
110
|
+
* and guessing "retry" silently drops the second copy of every conversation a
|
|
111
|
+
* file lists more than once.
|
|
112
|
+
*
|
|
113
|
+
* SENDING A FILE IN PAGES. Either give each row its own `row_ids` entry, and
|
|
114
|
+
* then chunking, ordering and subsets all stop mattering; or send one
|
|
115
|
+
* `import_id` with the `row_offset` each page starts at. Do not send back
|
|
116
|
+
* only the rows `notSavedRows` names unless you used `row_ids`: without them
|
|
117
|
+
* a shorter list moves every row after the gap and imports it again.
|
|
118
|
+
*/
|
|
119
|
+
importRows(params: LoopImportParams): Promise<LoopImportResult>;
|
|
120
|
+
/** List captured conversations, newest first. */
|
|
121
|
+
listTraces(params?: LoopTraceListParams): Promise<LoopTraceListResponse>;
|
|
122
|
+
/** Read one conversation, with every verdict recorded on it. */
|
|
123
|
+
getTrace(id: string): Promise<LoopTrace>;
|
|
124
|
+
/**
|
|
125
|
+
* Delete a conversation.
|
|
126
|
+
*
|
|
127
|
+
* It also leaves every training set built from it, including sets already
|
|
128
|
+
* exported -- so a customer asking you to delete a conversation genuinely
|
|
129
|
+
* removes it from what the model will learn next. Training runs that have
|
|
130
|
+
* already finished are not affected.
|
|
131
|
+
*/
|
|
132
|
+
deleteTrace(id: string): Promise<{
|
|
133
|
+
deleted: boolean;
|
|
134
|
+
trace_id: string;
|
|
135
|
+
}>;
|
|
136
|
+
/**
|
|
137
|
+
* Record a verdict on a conversation.
|
|
138
|
+
*
|
|
139
|
+
* Append-only: posting a second verdict does not replace the first. Two
|
|
140
|
+
* reviewers disagreeing about an answer is information worth keeping.
|
|
141
|
+
*
|
|
142
|
+
* `source` says who judged: `human`, `verifier` (a deterministic check --
|
|
143
|
+
* tests passed, schema valid), `judge` (a model grading a model), or
|
|
144
|
+
* `behavioural` (what the user did next). When several disagree, a human
|
|
145
|
+
* outranks a verifier, a verifier outranks a judge, and a judge outranks a
|
|
146
|
+
* behavioural hint.
|
|
147
|
+
*/
|
|
148
|
+
signal(traceId: string, params: LoopSignalParams): Promise<LoopSignal>;
|
|
149
|
+
/**
|
|
150
|
+
* Thumbs up: this answer was good. Shorthand for `signal(id, { verdict:
|
|
151
|
+
* 'accepted' })` — there is one stored vocabulary behind both, so a thumbs-up
|
|
152
|
+
* and an "accepted" can never disagree in your corpus.
|
|
153
|
+
*/
|
|
154
|
+
upvote(traceId: string, reason?: string): Promise<LoopSignal>;
|
|
155
|
+
/**
|
|
156
|
+
* Thumbs down: this answer was bad, and you have nothing better to offer.
|
|
157
|
+
*
|
|
158
|
+
* On its own this REMOVES the answer from training rather than teaching
|
|
159
|
+
* anything — there is no better version to learn. If you know what it should
|
|
160
|
+
* have said, `correct()` is worth far more.
|
|
161
|
+
*/
|
|
162
|
+
downvote(traceId: string, reason?: string): Promise<LoopSignal>;
|
|
163
|
+
/**
|
|
164
|
+
* The model was wrong; here is what it should have said.
|
|
165
|
+
*
|
|
166
|
+
* The most valuable feedback there is, because it produces BOTH halves of a
|
|
167
|
+
* preference pair from one action: the model learns your answer and learns to
|
|
168
|
+
* avoid its own.
|
|
169
|
+
*/
|
|
170
|
+
correct(traceId: string, answer: string, reason?: string): Promise<LoopSignal>;
|
|
171
|
+
/**
|
|
172
|
+
* Record the GOLD answer for this question — the reference, regardless of
|
|
173
|
+
* what the model happened to say.
|
|
174
|
+
*
|
|
175
|
+
* Different from `correct()` in a way that matters: a correction asserts the
|
|
176
|
+
* model was wrong, so the original becomes the rejected half of a pair. A gold
|
|
177
|
+
* answer asserts nothing about the model, so if it MATCHES what was said, no
|
|
178
|
+
* preference pair is invented — but SFT still learns it.
|
|
179
|
+
*
|
|
180
|
+
* It is the most reusable thing you can record: it trains SFT, forms a DPO
|
|
181
|
+
* pair when it differs from the answer, and stands in as the GRPO reference
|
|
182
|
+
* when you have not supplied a separate ground truth.
|
|
183
|
+
*/
|
|
184
|
+
gold(traceId: string, answer: string, reason?: string): Promise<LoopSignal>;
|
|
185
|
+
/** Every verdict on one conversation, oldest first. */
|
|
186
|
+
listSignals(traceId: string): Promise<LoopSignal[]>;
|
|
187
|
+
/**
|
|
188
|
+
* Submit an alternative answer to a prompt already captured.
|
|
189
|
+
*
|
|
190
|
+
* This is what makes preference learning scale. A human rewrite is the best
|
|
191
|
+
* signal there is and the least available -- somebody has to sit down and
|
|
192
|
+
* write it. Sample the model several times for the same prompt, score the
|
|
193
|
+
* samples, and a DPO pair falls out automatically: best becomes chosen,
|
|
194
|
+
* worst becomes rejected. Point a stronger model at the prompt instead and
|
|
195
|
+
* the same machinery does distillation.
|
|
196
|
+
*
|
|
197
|
+
* Score them, or they cannot pair: an unscored alternative says nothing about
|
|
198
|
+
* which answer anybody prefers. `score_source` is required alongside a score,
|
|
199
|
+
* because precedence between a verifier, a judge and a person is the whole
|
|
200
|
+
* reason we record who judged.
|
|
201
|
+
*
|
|
202
|
+
* A human correction always outranks any score.
|
|
203
|
+
*
|
|
204
|
+
* @example
|
|
205
|
+
* ```ts
|
|
206
|
+
* for (const sample of await sampleMyModel(prompt, 4)) {
|
|
207
|
+
* await client.loop.addCandidate(traceId, {
|
|
208
|
+
* completion: sample.text,
|
|
209
|
+
* score: await myVerifier(sample.text),
|
|
210
|
+
* score_source: 'verifier',
|
|
211
|
+
* });
|
|
212
|
+
* }
|
|
213
|
+
* ```
|
|
214
|
+
*/
|
|
215
|
+
addCandidate(traceId: string, params: LoopCandidateParams): Promise<LoopCandidate>;
|
|
216
|
+
/** Every alternative answer recorded for one prompt, oldest first. */
|
|
217
|
+
listCandidates(traceId: string): Promise<LoopCandidate[]>;
|
|
218
|
+
/**
|
|
219
|
+
* Label a conversation, so it can be selected later.
|
|
220
|
+
*
|
|
221
|
+
* An average over everything is the least useful thing to train on. A model
|
|
222
|
+
* weak at refunds is fixed with refund examples, and you can only select
|
|
223
|
+
* those if you said so when the conversation arrived.
|
|
224
|
+
*
|
|
225
|
+
* A label may name a `parent`, which is what makes a MASTER GROUP: labelling
|
|
226
|
+
* something `refunds` with parent `billing` makes it selectable as either,
|
|
227
|
+
* and selecting `billing` later gathers every child without you maintaining
|
|
228
|
+
* a list of them.
|
|
229
|
+
*
|
|
230
|
+
* Labels are lowercased and trimmed, so `Refunds` and `refunds` are one label.
|
|
231
|
+
*/
|
|
232
|
+
label(traceId: string, labels: string[], parent?: string, attributes?: Record<string, string>): Promise<LoopLabel[]>;
|
|
233
|
+
/**
|
|
234
|
+
* Set named dimensions on a conversation: `{ category: 'billing',
|
|
235
|
+
* language: 'es' }`.
|
|
236
|
+
*
|
|
237
|
+
* An upsert per dimension, so correcting `category` leaves `task` and any
|
|
238
|
+
* bare tags exactly where they were — no delete-then-add.
|
|
239
|
+
*/
|
|
240
|
+
setAttributes(traceId: string, attributes: Record<string, string>): Promise<LoopLabel[]>;
|
|
241
|
+
unlabel(traceId: string, label: string, key?: string): Promise<{
|
|
242
|
+
deleted: boolean;
|
|
243
|
+
key: string;
|
|
244
|
+
label: string;
|
|
245
|
+
}>;
|
|
246
|
+
/**
|
|
247
|
+
* Every label in the workspace, with how many conversations carry it and
|
|
248
|
+
* which master groups those conversations are in.
|
|
249
|
+
*
|
|
250
|
+
* TWO FIELDS ANSWER THE GROUP QUESTION, AND THEY ANSWER DIFFERENT ONES.
|
|
251
|
+
* `parent` is the group ALL of a label's conversations are in, and it is
|
|
252
|
+
* null the moment they disagree. `parents` is every group ANY of them are
|
|
253
|
+
* in, with how many of them are in each.
|
|
254
|
+
*
|
|
255
|
+
* Build a list of master groups out of `parents`. A label with 12
|
|
256
|
+
* conversations, 5 of them under `support`, reports `parent: null` and
|
|
257
|
+
* `parents: [{parent: 'support', traces: 5}]`, so code that reads only
|
|
258
|
+
* `parent` sees no group at all for it. The registry used to answer
|
|
259
|
+
* `parent: "support"` for all twelve, which named the group but put seven
|
|
260
|
+
* conversations in it that nobody had put there.
|
|
261
|
+
*/
|
|
262
|
+
listLabels(): Promise<LoopLabelCount[]>;
|
|
263
|
+
/**
|
|
264
|
+
* Build a training set from the feedback recorded so far.
|
|
265
|
+
*
|
|
266
|
+
* The three methods need genuinely different things, so one set cannot be
|
|
267
|
+
* reshaped into another afterwards:
|
|
268
|
+
*
|
|
269
|
+
* - `sft` -- answers you marked right, and answers you rewrote.
|
|
270
|
+
* - `dpo` -- answers you REWROTE, so a better and a worse version of the same
|
|
271
|
+
* reply exist. Nothing else produces a pair.
|
|
272
|
+
* - `grpo` -- answers with a value or fact they can be checked against, or
|
|
273
|
+
* a judge rubric, which scores answers the run has not written yet.
|
|
274
|
+
* - `kto` -- anything carrying a yes or a no, INCLUDING a thumbs-down with
|
|
275
|
+
* nothing written. That row trains nothing under the other three methods,
|
|
276
|
+
* which is why this one exists: it is the feedback people actually give.
|
|
277
|
+
*
|
|
278
|
+
* `holdout_percent` holds back your most recent work rather than a random
|
|
279
|
+
* slice, so the evaluation measures whether the model generalised instead of
|
|
280
|
+
* memorised the same week.
|
|
281
|
+
*
|
|
282
|
+
* The result always reports `rejected_counts`: why rows were left out. A
|
|
283
|
+
* small set with a reason is useful; a small set without one is just alarming.
|
|
284
|
+
*/
|
|
285
|
+
createDataset(params: LoopDatasetCreateParams): Promise<LoopDataset>;
|
|
286
|
+
/** List training sets, newest first. */
|
|
287
|
+
/**
|
|
288
|
+
* Stand up a rule that builds a set whenever enough new work exists.
|
|
289
|
+
*
|
|
290
|
+
* The manual path asks somebody to notice that enough conversations have
|
|
291
|
+
* been reviewed, remember which filters describe the slice they want, and
|
|
292
|
+
* press build — every time. A rule is that instruction, stored.
|
|
293
|
+
*
|
|
294
|
+
* `spec` is the same selection `createDataset` takes and is replayed
|
|
295
|
+
* verbatim, so an automatic set is identical to a hand-made one.
|
|
296
|
+
*
|
|
297
|
+
* `min_new_rows` counts only work reviewed SINCE THE LAST BUILD. Counting
|
|
298
|
+
* the whole corpus would fire the rule every interval forever, because a
|
|
299
|
+
* total that has crossed a threshold stays across it.
|
|
300
|
+
*/
|
|
301
|
+
createBuildRule(params: LoopBuildRuleParams): Promise<LoopBuildRule>;
|
|
302
|
+
/**
|
|
303
|
+
* Every build rule, with what each one last did and why.
|
|
304
|
+
*
|
|
305
|
+
* `last_reason` is the field worth reading: a rule quiet because it is
|
|
306
|
+
* waiting looks exactly like one quiet because it is broken.
|
|
307
|
+
*/
|
|
308
|
+
listBuildRules(): Promise<LoopBuildRule[]>;
|
|
309
|
+
/** Stop a standing build rule. Sets it already produced are untouched. */
|
|
310
|
+
deleteBuildRule(id: string): Promise<{
|
|
311
|
+
deleted: boolean;
|
|
312
|
+
}>;
|
|
313
|
+
listDatasets(params?: LoopDatasetListParams): Promise<LoopDataset[]>;
|
|
314
|
+
/** Read one training set and its curation report. */
|
|
315
|
+
getDataset(id: string): Promise<LoopDataset>;
|
|
316
|
+
/**
|
|
317
|
+
* Page through the rows a training set actually contains.
|
|
318
|
+
*
|
|
319
|
+
* Worth reading before you spend money on a run: each row carries the address
|
|
320
|
+
* of the conversation it was built from.
|
|
321
|
+
*/
|
|
322
|
+
listDatasetItems(id: string, params?: {
|
|
323
|
+
limit?: number;
|
|
324
|
+
offset?: number;
|
|
325
|
+
}): Promise<LoopDatasetItem[]>;
|
|
326
|
+
/**
|
|
327
|
+
* Download a training set as JSONL -- one training row per line, the format
|
|
328
|
+
* every trainer in this space reads.
|
|
329
|
+
*
|
|
330
|
+
* `split` defaults to the training rows; pass `holdout` for the slice held
|
|
331
|
+
* back, or `all` for both.
|
|
332
|
+
*/
|
|
333
|
+
downloadDataset(id: string, split?: 'train' | 'holdout' | 'all'): Promise<string>;
|
|
334
|
+
/**
|
|
335
|
+
* Delete a training set.
|
|
336
|
+
*
|
|
337
|
+
* The conversations it was built from are untouched -- a set is a selection,
|
|
338
|
+
* and discarding the selection must not discard the evidence.
|
|
339
|
+
*/
|
|
340
|
+
deleteDataset(id: string): Promise<{
|
|
341
|
+
deleted: boolean;
|
|
342
|
+
dataset_id: string;
|
|
343
|
+
}>;
|
|
344
|
+
/**
|
|
345
|
+
* Write a rule that scores answers without a person.
|
|
346
|
+
*
|
|
347
|
+
* Human review is the most trustworthy feedback and the least available. A
|
|
348
|
+
* grader is written once and applied to every answer afterwards: did it
|
|
349
|
+
* contain the required phrase, did it parse as the schema you asked for, is
|
|
350
|
+
* the number within tolerance of the known answer, did it call the function
|
|
351
|
+
* it should have.
|
|
352
|
+
*
|
|
353
|
+
* Every kind here is DETERMINISTIC -- no model call, no network. That is why
|
|
354
|
+
* a verifier outranks a judge when they disagree: it cannot be flattered and
|
|
355
|
+
* it cannot drift between runs.
|
|
356
|
+
*
|
|
357
|
+
* `weight` is the multiplier: a criterion that matters twice as much gets
|
|
358
|
+
* twice the weight, and the combined score is the weighted mean over the
|
|
359
|
+
* rules that actually applied.
|
|
360
|
+
*
|
|
361
|
+
* `matches_gold` needs no `expected`: it compares each answer to the gold
|
|
362
|
+
* answer recorded on that same conversation, or the human correction when
|
|
363
|
+
* there is no gold, and scores sampled alternatives against the same gold.
|
|
364
|
+
* A conversation with neither is skipped, not failed, so one rule checks
|
|
365
|
+
* every labelled question without punishing the unlabelled ones. An
|
|
366
|
+
* optional `tolerance` widens the match when both sides are bare numbers.
|
|
367
|
+
*
|
|
368
|
+
* A caution worth knowing before you write a set: a rule made only of
|
|
369
|
+
* `forbidden` phrases is satisfied VACUOUSLY by an answer that says nothing.
|
|
370
|
+
* Pair it with a `required` phrase, or you are rewarding silence.
|
|
371
|
+
*/
|
|
372
|
+
createGrader(params: LoopGraderParams): Promise<LoopGrader>;
|
|
373
|
+
/** Every rule this workspace has written. */
|
|
374
|
+
listGraders(): Promise<LoopGrader[]>;
|
|
375
|
+
deleteGrader(id: string): Promise<{
|
|
376
|
+
deleted: boolean;
|
|
377
|
+
grader_id: string;
|
|
378
|
+
}>;
|
|
379
|
+
/**
|
|
380
|
+
* Apply this workspace's rules to one captured answer AND to every
|
|
381
|
+
* alternative sampled for it.
|
|
382
|
+
*
|
|
383
|
+
* Grading both in one pass is the point: scoring only the original gives a
|
|
384
|
+
* verdict, while scoring the samples as well gives the preference pair. Sample
|
|
385
|
+
* your model, call this, and you have DPO data with nobody reading anything.
|
|
386
|
+
*
|
|
387
|
+
* A run where no rule could apply writes NOTHING -- no verdict, no scores.
|
|
388
|
+
* Recording a zero that no rule produced would poison curation with a
|
|
389
|
+
* judgement nobody reached, so `applied: 0` is reported instead.
|
|
390
|
+
*/
|
|
391
|
+
grade(traceId: string): Promise<LoopGradeResult>;
|
|
392
|
+
/** Every source this workspace has configured. */
|
|
393
|
+
listConfigs(): Promise<LoopConfig[]>;
|
|
394
|
+
/**
|
|
395
|
+
* Read the capture setting for one source. A source nobody has configured
|
|
396
|
+
* reads back as disabled rather than missing, because "we are not recording
|
|
397
|
+
* this" is the honest answer.
|
|
398
|
+
*/
|
|
399
|
+
getConfig(source: string): Promise<LoopConfig>;
|
|
400
|
+
/**
|
|
401
|
+
* Turn capture on or off for a source, and choose how long it is kept.
|
|
402
|
+
*
|
|
403
|
+
* This is the consent decision the whole feature rests on: nothing is
|
|
404
|
+
* recorded until it is made, and the row remembers who made it and when.
|
|
405
|
+
* `sample_rate` below 1 records a deterministic fraction, chosen so that every
|
|
406
|
+
* turn of one conversation is captured or skipped together -- half a
|
|
407
|
+
* conversation makes training rows with holes in them.
|
|
408
|
+
*/
|
|
409
|
+
setConfig(source: string, params: LoopConfigParams): Promise<LoopConfig>;
|
|
410
|
+
/**
|
|
411
|
+
* How much there is, and how much of it each method could actually use.
|
|
412
|
+
*
|
|
413
|
+
* The `ready` numbers are upper bounds: duplicate questions are folded into
|
|
414
|
+
* one row while a set is built, so the finished count can be lower.
|
|
415
|
+
*/
|
|
416
|
+
stats(): Promise<LoopStats>;
|
|
417
|
+
/** Write a rubric: what makes an answer good here, and what to score. */
|
|
418
|
+
createJudge(params: LoopJudgeParams): Promise<LoopJudge>;
|
|
419
|
+
listJudges(): Promise<LoopJudge[]>;
|
|
420
|
+
/**
|
|
421
|
+
* Rewrite a rubric. Runs already recorded keep the instructions they used:
|
|
422
|
+
* a verdict has to keep meaning what it meant when it was given.
|
|
423
|
+
*/
|
|
424
|
+
updateJudge(judgeId: string, params: LoopJudgeParams): Promise<LoopJudge>;
|
|
425
|
+
/**
|
|
426
|
+
* Retire a rubric. Its runs go with it; the verdicts they wrote stay. A score
|
|
427
|
+
* is evidence about an answer, and it does not stop being true because the
|
|
428
|
+
* rubric was retired.
|
|
429
|
+
*/
|
|
430
|
+
deleteJudge(judgeId: string): Promise<void>;
|
|
431
|
+
/**
|
|
432
|
+
* Open a run: select the conversations now, and freeze the rubric onto them.
|
|
433
|
+
*
|
|
434
|
+
* The selection is fixed at this moment on purpose. A run whose selection is
|
|
435
|
+
* a live query silently grows as conversations arrive, so "we scored the
|
|
436
|
+
* refunds slice" becomes a claim about a set that no longer exists.
|
|
437
|
+
*/
|
|
438
|
+
startRun(judgeId: string, limit?: number): Promise<LoopJudgeRun>;
|
|
439
|
+
listRuns(judgeId?: string): Promise<LoopJudgeRun[]>;
|
|
440
|
+
getRun(runId: string): Promise<LoopJudgeRun>;
|
|
441
|
+
/**
|
|
442
|
+
* Take the next conversations to score, each rendered into a ready-to-send
|
|
443
|
+
* prompt. Nothing is marked taken, so a caller that dies half way loses
|
|
444
|
+
* nothing: ask again and the same items come back.
|
|
445
|
+
*/
|
|
446
|
+
takeWork(runId: string, limit?: number): Promise<LoopJudgeWork>;
|
|
447
|
+
/**
|
|
448
|
+
* What happened to each conversation in a run, and why.
|
|
449
|
+
*
|
|
450
|
+
* `takeWork` hands out what is still PENDING, so a finished run answers it
|
|
451
|
+
* with an empty list. This answers with every item and the outcome on it:
|
|
452
|
+
* `status`, `scored_at`, and `error` — the reason the caller gave for an
|
|
453
|
+
* item it could not score, which is where a wrong model slug or a refused
|
|
454
|
+
* key actually shows up. A run that ends "scored 0, failed 3" is otherwise a
|
|
455
|
+
* number with no detail behind it.
|
|
456
|
+
*
|
|
457
|
+
* Pass `status: 'failed'` for the usual question. At most 500 items come
|
|
458
|
+
* back at a time; when `has_more` is true, call again with the `next_offset`
|
|
459
|
+
* from the reply.
|
|
460
|
+
*/
|
|
461
|
+
listRunItems(runId: string, opts?: {
|
|
462
|
+
status?: LoopJudgeRunItemStatus;
|
|
463
|
+
limit?: number;
|
|
464
|
+
offset?: number;
|
|
465
|
+
}): Promise<LoopJudgeRunItems>;
|
|
466
|
+
/**
|
|
467
|
+
* Hand the scores back. A verdict for a conversation outside this run, or
|
|
468
|
+
* scoring something the rubric never asked for, is refused and reported in
|
|
469
|
+
* `rejected` rather than silently dropped.
|
|
470
|
+
*
|
|
471
|
+
* Pass `finish` to close the run in the same call once you have nothing left
|
|
472
|
+
* to send. To close a run WITHOUT scores, call `stopRun`: an empty verdict
|
|
473
|
+
* list with `finish` does the same thing and reads like a mistake.
|
|
474
|
+
*/
|
|
475
|
+
postVerdicts(runId: string, verdicts: LoopJudgeVerdict[], finish?: boolean): Promise<LoopJudgeVerdictResult>;
|
|
476
|
+
/**
|
|
477
|
+
* Close a run that has not finished.
|
|
478
|
+
*
|
|
479
|
+
* A judge may have one open run of its own at a time, so an open run BLOCKS
|
|
480
|
+
* the next one, and the runs that most need closing are the ones nobody can
|
|
481
|
+
* wait out: a run the agent has parked on an empty balance or a refused key
|
|
482
|
+
* stays open until the reason is fixed or somebody stops it.
|
|
483
|
+
*
|
|
484
|
+
* Nothing is deleted. Every verdict already recorded stays recorded, the
|
|
485
|
+
* counters keep saying how much of the selection was covered, and the
|
|
486
|
+
* conversations the run was holding are free for the next one. The run ends
|
|
487
|
+
* as `stopped` rather than `done`, so an interrupted pass and a completed
|
|
488
|
+
* one do not read alike.
|
|
489
|
+
*
|
|
490
|
+
* A run that finished on its own is not rewritten: stopping one answers 409.
|
|
491
|
+
*
|
|
492
|
+
* Returns the stopped run itself, like `startRun` and `getRun`, not the
|
|
493
|
+
* `{run}` envelope the service sends. Every other single-run method in this
|
|
494
|
+
* class unwraps, and `Loop.stop_run` in the Python SDK does too, so a
|
|
495
|
+
* caller who wrote `(await loop.stopRun(id)).status` against one of its
|
|
496
|
+
* siblings is right here as well.
|
|
497
|
+
*/
|
|
498
|
+
stopRun(runId: string): Promise<LoopJudgeRun>;
|
|
499
|
+
/**
|
|
500
|
+
* Can an agent run here, is one running, and is it on for this workspace.
|
|
501
|
+
* `available` is about the environment; `online` about the worker;
|
|
502
|
+
* `credential` about this workspace.
|
|
503
|
+
*/
|
|
504
|
+
agentStatus(): Promise<LoopAgentStatus>;
|
|
505
|
+
/**
|
|
506
|
+
* Turn the agent on: mints the workspace's managed serverless key and
|
|
507
|
+
* resumes the automatic judges a previous turn-off paused. After this,
|
|
508
|
+
* automatic judges and sample runs make model calls billed to the
|
|
509
|
+
* workspace. Idempotent.
|
|
510
|
+
*
|
|
511
|
+
* A judge whose model the serving gateway will not route is NOT resumed:
|
|
512
|
+
* making it automatic would buy a run that fails every conversation. It
|
|
513
|
+
* stays paused, `judges_still_paused` counts those, and each one carries
|
|
514
|
+
* `auto_pause_reason` saying so. Point it at a model that is served and
|
|
515
|
+
* turn the agent on again.
|
|
516
|
+
*
|
|
517
|
+
* `monthly_spend_cap_cents` caps what that key may spend on model calls in
|
|
518
|
+
* a calendar month. Omitted = no cap on a fresh key, and an existing cap is
|
|
519
|
+
* left as it is; sent while the agent is already on, it moves the cap on
|
|
520
|
+
* the existing key without minting a new one. When the cap is reached the
|
|
521
|
+
* agent's model calls are refused until next month and its runs pause with
|
|
522
|
+
* that reason.
|
|
523
|
+
*/
|
|
524
|
+
enableAgent(opts?: {
|
|
525
|
+
monthly_spend_cap_cents?: number;
|
|
526
|
+
}): Promise<LoopAgentCredential>;
|
|
527
|
+
/**
|
|
528
|
+
* Turn the agent off: revokes its key and pauses every automatic judge in
|
|
529
|
+
* the workspace (`judges_paused`; each judge shows `auto_paused`). Turning
|
|
530
|
+
* it back on resumes exactly those judges. Open runs stop where they are
|
|
531
|
+
* and continue if it is turned back on. Nothing already scored or written
|
|
532
|
+
* is removed.
|
|
533
|
+
*/
|
|
534
|
+
disableAgent(): Promise<{
|
|
535
|
+
revoked: boolean;
|
|
536
|
+
judges_paused: number;
|
|
537
|
+
message: string;
|
|
538
|
+
}>;
|
|
539
|
+
/**
|
|
540
|
+
* Ask the agent to write `n` alternative answers to each conversation in a
|
|
541
|
+
* slice with `model`, score each with `judge_id`, and store them as
|
|
542
|
+
* candidates. This is how preference pairs are made without a person
|
|
543
|
+
* writing each one: the curation pass pairs the best sample against the
|
|
544
|
+
* worst wherever the gap is real.
|
|
545
|
+
*
|
|
546
|
+
* Cost: up to `n` calls to write plus `n` to judge, per conversation, at the
|
|
547
|
+
* workspace's serverless rate; `selection.sample` caps the conversations
|
|
548
|
+
* (max 200) and `n` is capped at 8. Opening a run turns the agent on if it
|
|
549
|
+
* is off.
|
|
550
|
+
*/
|
|
551
|
+
createSampleRun(params: LoopSampleRunParams): Promise<LoopSampleRun>;
|
|
552
|
+
listSampleRuns(): Promise<LoopSampleRun[]>;
|
|
553
|
+
getSampleRun(runId: string): Promise<LoopSampleRun>;
|
|
554
|
+
/**
|
|
555
|
+
* Price a training rule before anyone agrees to it. Writes nothing.
|
|
556
|
+
*
|
|
557
|
+
* Takes the create body without `accept_terms` and answers with the estimate
|
|
558
|
+
* a member has to see first: the pinned model revision, the worst hourly
|
|
559
|
+
* price each GPU ladder can reach, how many hours each ceiling buys, any
|
|
560
|
+
* refusals that would stop a create, the API keys that cannot follow a
|
|
561
|
+
* cutover because they lack `deployments:read`, and `terms_text` -- the
|
|
562
|
+
* exact sentence to show, with real figures in it.
|
|
563
|
+
*
|
|
564
|
+
* Send the `terms_version` it returns back in `createTrainingRule`. Read the
|
|
565
|
+
* figures out of this response rather than inventing ceilings of your own.
|
|
566
|
+
*
|
|
567
|
+
* `valid: false` with an EMPTY `refusals` list is not nothing: check
|
|
568
|
+
* `unreachable`, which names every peer the platform could not reach. The
|
|
569
|
+
* rule is savable in that state and would be paused, but the estimate around
|
|
570
|
+
* it is not trustworthy -- both `worst_hourly_*` are `0`, and so are both
|
|
571
|
+
* `max_*_hours` beside them -- so do not quote those figures to anyone.
|
|
572
|
+
* `model_revision` is empty unless training-service actually pinned a
|
|
573
|
+
* commit; test `unreachable.length > 0` to tell a degraded pass from a
|
|
574
|
+
* refusal.
|
|
575
|
+
*/
|
|
576
|
+
preflightTrainingRule(params: TrainingRulePreflightRequest): Promise<TrainingRulePreflight>;
|
|
577
|
+
/**
|
|
578
|
+
* Create a training rule and record the consent that pays for it.
|
|
579
|
+
*
|
|
580
|
+
* ACCEPTING THE TERMS AUTHORISES SPENDING FROM YOUR WALLET WHILE YOU ARE NOT
|
|
581
|
+
* PRESENT. From here on the platform may, on its own schedule and without
|
|
582
|
+
* asking again, build a training set, run a training job on rented GPUs,
|
|
583
|
+
* book a second deployment to compare against the one serving your traffic,
|
|
584
|
+
* and pay for the model calls that judge the two. Those charges come out of
|
|
585
|
+
* the wallet of the member whose credentials make this call, up to the
|
|
586
|
+
* ceilings in `money`, and they keep recurring for as long as the rule is
|
|
587
|
+
* enabled. If `promotion.auto_promote` is set, the platform will also
|
|
588
|
+
* re-point your public handle at the new model with nobody reviewing it.
|
|
589
|
+
*
|
|
590
|
+
* `accept_terms` is therefore required, and this method refuses to send the
|
|
591
|
+
* request without it rather than letting the server decide. Call
|
|
592
|
+
* {@link preflightTrainingRule} first, show the person whose wallet pays the
|
|
593
|
+
* `terms_text` and the figures it returns, get an explicit yes, and send the
|
|
594
|
+
* `terms_version` they were shown. Do not invent ceilings or price caps on
|
|
595
|
+
* their behalf.
|
|
596
|
+
*
|
|
597
|
+
* @example
|
|
598
|
+
* ```ts
|
|
599
|
+
* const estimate = await client.loop.preflightTrainingRule(draft);
|
|
600
|
+
* // show estimate.terms_text and the ceilings to the member, get a yes
|
|
601
|
+
* const rule = await client.loop.createTrainingRule({
|
|
602
|
+
* ...draft,
|
|
603
|
+
* accept_terms: { terms_version: estimate.terms_version },
|
|
604
|
+
* });
|
|
605
|
+
* ```
|
|
606
|
+
*/
|
|
607
|
+
createTrainingRule(params: TrainingRuleCreateRequest): Promise<TrainingRule>;
|
|
608
|
+
/**
|
|
609
|
+
* Every training rule in the workspace, with what each one last did and why.
|
|
610
|
+
*
|
|
611
|
+
* `last_reason` and `paused_reason` are the fields worth reading: a rule
|
|
612
|
+
* quiet because it is waiting looks exactly like one quiet because its
|
|
613
|
+
* consent went stale.
|
|
614
|
+
*/
|
|
615
|
+
listTrainingRules(params?: TrainingRuleListParams): Promise<TrainingRuleListResponse>;
|
|
616
|
+
/**
|
|
617
|
+
* Read one rule with its recent runs, what it has spent this month, and how
|
|
618
|
+
* far its judge agrees with your own reviewers.
|
|
619
|
+
*
|
|
620
|
+
* Returned whole rather than unwrapped to the rule: `month_spent_cents` is
|
|
621
|
+
* the number that says whether the monthly ceiling is about to stop it.
|
|
622
|
+
*/
|
|
623
|
+
getTrainingRule(id: string): Promise<TrainingRuleResponse>;
|
|
624
|
+
/**
|
|
625
|
+
* Edit or pause a rule. Absent means unchanged; a null clears the fields
|
|
626
|
+
* that can be cleared.
|
|
627
|
+
*
|
|
628
|
+
* A money-bearing change -- a ceiling, a model, a GPU ladder, the promotion
|
|
629
|
+
* policy -- bumps `revision`, clears the recorded consent and STOPS the rule
|
|
630
|
+
* firing until someone accepts the new amounts. The reply says so in
|
|
631
|
+
* `consent_required`, and carries a fresh `preflight` with the new figures.
|
|
632
|
+
* Pass `expected_revision` to be refused with `REVISION_MISMATCH` rather
|
|
633
|
+
* than overwrite an edit somebody else made in the meantime.
|
|
634
|
+
*/
|
|
635
|
+
updateTrainingRule(id: string, params: TrainingRuleUpdateRequest): Promise<TrainingRuleMutationResponse>;
|
|
636
|
+
/**
|
|
637
|
+
* Accept the rule's current terms, so it may fire again.
|
|
638
|
+
*
|
|
639
|
+
* ACCEPTING THE TERMS AUTHORISES SPENDING FROM YOUR WALLET WHILE YOU ARE NOT
|
|
640
|
+
* PRESENT, on the amounts as they stand right now. This is the same
|
|
641
|
+
* authorisation {@link createTrainingRule} records, given again because a
|
|
642
|
+
* money-bearing edit cleared the old one: training, a candidate deployment
|
|
643
|
+
* and the judge's model calls are charged to the wallet of the member making
|
|
644
|
+
* this call, up to the rule's ceilings, every time it fires.
|
|
645
|
+
*
|
|
646
|
+
* Show the member the current `terms_text` from a fresh
|
|
647
|
+
* {@link preflightTrainingRule} or from the `preflight` on the update reply,
|
|
648
|
+
* and send the `revision` those figures belong to. A stale revision is
|
|
649
|
+
* refused with `409`, which is the point: it means the amounts moved again
|
|
650
|
+
* after they were read.
|
|
651
|
+
*/
|
|
652
|
+
consentTrainingRule(id: string, params: TrainingRuleConsentRequest): Promise<TrainingRule>;
|
|
653
|
+
/**
|
|
654
|
+
* Fire a rule now, without waiting for its cadence.
|
|
655
|
+
*
|
|
656
|
+
* Bypasses the schedule and `min_new_rows` only. The row floors, the money
|
|
657
|
+
* ceilings and the consent all still apply, so this can answer `409
|
|
658
|
+
* CONSENT_REQUIRED` or `422 NOT_ENOUGH_ROWS` with the counts it needed.
|
|
659
|
+
*/
|
|
660
|
+
runTrainingRule(id: string): Promise<TrainingRun>;
|
|
661
|
+
/**
|
|
662
|
+
* Delete a rule. An active run is cancelled; runs that already finished, and
|
|
663
|
+
* anything already promoted, are kept.
|
|
664
|
+
*/
|
|
665
|
+
deleteTrainingRule(id: string): Promise<TrainingRuleDeleteResponse>;
|
|
666
|
+
/**
|
|
667
|
+
* Edit or pause a build rule -- the standing instruction that assembles the
|
|
668
|
+
* training set a training rule then trains on.
|
|
669
|
+
*
|
|
670
|
+
* Changing `spec.deployment_id` while an enabled training rule owns this
|
|
671
|
+
* build rule is refused with `409`: it would silently retrain the next model
|
|
672
|
+
* on a different source's conversations.
|
|
673
|
+
*/
|
|
674
|
+
updateBuildRule(id: string, params: LoopBuildRuleUpdateParams): Promise<LoopBuildRule>;
|
|
675
|
+
/** Firings, newest first. Filter by rule or by the state they are sitting in. */
|
|
676
|
+
listTrainingRuns(params?: TrainingRunListParams): Promise<TrainingRunListResponse>;
|
|
677
|
+
/**
|
|
678
|
+
* One run with its timeline, the addresses of everything it created, and
|
|
679
|
+
* what you may do with it right now.
|
|
680
|
+
*
|
|
681
|
+
* `timeline` is written in the same transaction as each state change, so it
|
|
682
|
+
* is the record of what actually happened rather than a reconstruction.
|
|
683
|
+
* `available_actions` is the honest answer to "can I promote this": a button
|
|
684
|
+
* that cannot work should never be offered.
|
|
685
|
+
*/
|
|
686
|
+
getTrainingRun(id: string): Promise<TrainingRunResponse>;
|
|
687
|
+
/**
|
|
688
|
+
* Promote the candidate by hand: re-point the public handle at the new model.
|
|
689
|
+
*
|
|
690
|
+
* This changes what answers your customers. Pass `expected_revision` to be
|
|
691
|
+
* refused rather than promote against a consent that moved underneath the
|
|
692
|
+
* decision, and `force` only to promote a comparison the evaluation called
|
|
693
|
+
* inconclusive -- without it that case is refused with
|
|
694
|
+
* `INCONCLUSIVE_REQUIRES_FORCE`.
|
|
695
|
+
*/
|
|
696
|
+
promoteTrainingRun(id: string, params?: TrainingRunPromoteRequest): Promise<TrainingRunActionResponse>;
|
|
697
|
+
/** Retire the candidate. What serves your traffic does not change. */
|
|
698
|
+
rejectTrainingRun(id: string, params?: TrainingRunRejectRequest): Promise<TrainingRunActionResponse>;
|
|
699
|
+
/**
|
|
700
|
+
* Undo a promotion, inside the rollback window the run reports in
|
|
701
|
+
* `rollback_available_until` (30 days).
|
|
702
|
+
*
|
|
703
|
+
* The reply carries `serving`: this is the one action besides promotion that
|
|
704
|
+
* changes what answers your traffic, so what it was put back to is returned
|
|
705
|
+
* rather than left to be looked up.
|
|
706
|
+
*/
|
|
707
|
+
rollbackTrainingRun(id: string, params?: TrainingRunRollbackRequest): Promise<TrainingRunActionResponse>;
|
|
708
|
+
/**
|
|
709
|
+
* Stop a run that is still moving. Work already paid for is still billed --
|
|
710
|
+
* cancelling a training job does not refund the hours it burned.
|
|
711
|
+
*/
|
|
712
|
+
cancelTrainingRun(id: string, params?: TrainingRunCancelRequest): Promise<TrainingRunActionResponse>;
|
|
713
|
+
/**
|
|
714
|
+
* The comparison behind a verdict: both models on the same held-out rows,
|
|
715
|
+
* with identical decoding, judge and grader names resolved.
|
|
716
|
+
*
|
|
717
|
+
* Read `warnings` before you read `win_rate`. A win rate over a holdout too
|
|
718
|
+
* small to mean anything, or one measured by a judge that disagrees with
|
|
719
|
+
* your own reviewers, is reported with the warning that says so rather than
|
|
720
|
+
* withheld. `margin_used` is the thresholds this verdict was measured
|
|
721
|
+
* against, frozen with the report, so a later policy change cannot rewrite
|
|
722
|
+
* what a past decision meant.
|
|
723
|
+
*/
|
|
724
|
+
getEvaluation(id: string): Promise<Evaluation>;
|
|
725
|
+
/**
|
|
726
|
+
* The paired conversations behind the numbers: one prompt, both answers, the
|
|
727
|
+
* scores each earned, and which won.
|
|
728
|
+
*
|
|
729
|
+
* Filter by `winner` to read the losses first, which is where a verdict is
|
|
730
|
+
* actually checked. `limit` is capped at 100 by the service.
|
|
731
|
+
*/
|
|
732
|
+
listEvaluationItems(id: string, params?: EvaluationItemListParams): Promise<EvaluationItemsResponse>;
|
|
733
|
+
/**
|
|
734
|
+
* How far a judge agrees with your own reviewers, over the conversations
|
|
735
|
+
* both have scored.
|
|
736
|
+
*
|
|
737
|
+
* This is a gate, not a badge: a rule's `min_judge_agreement` refuses to
|
|
738
|
+
* promote on the word of a judge that does not agree with the people whose
|
|
739
|
+
* product it is. `enough_pairs` is the field to read first -- "not enough
|
|
740
|
+
* reviewer overlap yet" is an answer, and 100% of two pairs is not.
|
|
741
|
+
*/
|
|
742
|
+
getJudgeAgreement(judgeId: string, params?: JudgeAgreementParams): Promise<JudgeAgreement>;
|
|
743
|
+
/**
|
|
744
|
+
* Pin a fixed set of conversations, with a judge frozen beside them, and
|
|
745
|
+
* measure every future run against it.
|
|
746
|
+
*
|
|
747
|
+
* The comparison answers "is this candidate better than what serves today".
|
|
748
|
+
* It cannot answer "is my model getting better", because its rows, its
|
|
749
|
+
* opponent and its judge all move between runs. A benchmark is the other
|
|
750
|
+
* instrument: the same conversations, the same oracle, the same decoding,
|
|
751
|
+
* replayed against both models on every run that finishes its comparison,
|
|
752
|
+
* and reported as two absolute numbers on a scale that does not move.
|
|
753
|
+
*
|
|
754
|
+
* 10 to 200 conversations. A named conversation with nothing to ask a model
|
|
755
|
+
* is refused rather than skipped, because pinning 47 of the 50 you chose is
|
|
756
|
+
* the set being wrong from the first day and you would never find out.
|
|
757
|
+
*
|
|
758
|
+
* It raises no amount you have already agreed to. The replay's calls come
|
|
759
|
+
* out of the rule's existing `eval_ceiling_cents`, and
|
|
760
|
+
* `per_run_ceiling_cents` can only lower what is spent inside that.
|
|
761
|
+
*/
|
|
762
|
+
createBenchmark(params: BenchmarkCreateParams): Promise<Benchmark>;
|
|
763
|
+
/** Your benchmarks, newest first. Filter by `status` to hide retired ones. */
|
|
764
|
+
listBenchmarks(params?: BenchmarkListParams): Promise<BenchmarkListResponse>;
|
|
765
|
+
/**
|
|
766
|
+
* One benchmark: the frozen oracle, the scoring rule and the fingerprint of
|
|
767
|
+
* the set.
|
|
768
|
+
*
|
|
769
|
+
* `items_digest` is recomputed from the rows and compared at the start of
|
|
770
|
+
* every replay, so "the set cannot drift" is something you can check rather
|
|
771
|
+
* than something the platform promises.
|
|
772
|
+
*/
|
|
773
|
+
getBenchmark(id: string): Promise<Benchmark>;
|
|
774
|
+
/**
|
|
775
|
+
* The pinned conversations, paged. `limit` is capped at 100 by the service.
|
|
776
|
+
*
|
|
777
|
+
* `source_trace_id` and `source_trace_url` come back null once the
|
|
778
|
+
* conversation a row was copied from has been deleted. The row itself stays
|
|
779
|
+
* and every number already measured against it stays exactly as comparable
|
|
780
|
+
* as it was: retention cannot shrink a benchmark.
|
|
781
|
+
*/
|
|
782
|
+
listBenchmarkItems(id: string, params?: BenchmarkItemListParams): Promise<BenchmarkItemsResponse>;
|
|
783
|
+
/**
|
|
784
|
+
* The trend: every score this benchmark has produced, newest first, each
|
|
785
|
+
* beside the checkpoint that produced it and what you then did about it.
|
|
786
|
+
*
|
|
787
|
+
* EVERY REPLAY IS A POINT, including the ones with no numbers on them. A
|
|
788
|
+
* series that silently dropped the runs where the benchmark could not be
|
|
789
|
+
* scored would read as an unbroken line and not be one, so those points come
|
|
790
|
+
* back with `candidate_score` null and `status_reason` saying why.
|
|
791
|
+
*/
|
|
792
|
+
getBenchmarkHistory(id: string, params?: BenchmarkHistoryParams): Promise<BenchmarkHistoryResponse>;
|
|
793
|
+
/**
|
|
794
|
+
* Stop replaying a benchmark, and detach it from every rule that names it.
|
|
795
|
+
*
|
|
796
|
+
* Nothing measured is removed. There is no delete and no update on purpose:
|
|
797
|
+
* the series of numbers measured against a set is what a benchmark is for,
|
|
798
|
+
* so an edit would make everything before it incomparable with everything
|
|
799
|
+
* after it and a delete would throw the series away. Changing the set means
|
|
800
|
+
* pinning a new benchmark, and retiring the old one releases its name.
|
|
801
|
+
*
|
|
802
|
+
* Retiring twice is somebody pressing a button twice: the second call
|
|
803
|
+
* answers with the retired benchmark rather than a refusal.
|
|
804
|
+
*/
|
|
805
|
+
retireBenchmark(id: string, params?: BenchmarkRetireParams): Promise<Benchmark>;
|
|
806
|
+
/**
|
|
807
|
+
* One replay in full: both absolute scores, the difference between them on
|
|
808
|
+
* the same set, the per-dimension breakdown and what it cost.
|
|
809
|
+
*
|
|
810
|
+
* Read `status` before you read the scores. A replay that did not score
|
|
811
|
+
* every pinned conversation publishes no score at all -- all three of
|
|
812
|
+
* `candidate_score`, `incumbent_score` and `score_delta` are null and
|
|
813
|
+
* `status_reason` is the sentence that explains it -- because a mean over
|
|
814
|
+
* whichever conversations happened to succeed is a measurement of a
|
|
815
|
+
* different set, which is the exact defect a standing benchmark removes.
|
|
816
|
+
*/
|
|
817
|
+
getBenchmarkRun(id: string): Promise<BenchmarkRun>;
|
|
818
|
+
/**
|
|
819
|
+
* Attach a benchmark to a rule, or send `{ benchmark_id: null }` to detach
|
|
820
|
+
* it.
|
|
821
|
+
*
|
|
822
|
+
* Its own route rather than a field on the rule body, because it is a
|
|
823
|
+
* decision to replay a fixed set on every future run of this rule for as
|
|
824
|
+
* long as it stands, and its refusals -- retired, or belonging to another
|
|
825
|
+
* workspace -- are about the benchmark rather than about the rule.
|
|
826
|
+
*
|
|
827
|
+
* It does not invalidate consent and the reply says so: attaching raises
|
|
828
|
+
* neither the amount set aside for judge calls nor the amount set aside for
|
|
829
|
+
* keeping the new model available, so nobody is asked to read the same
|
|
830
|
+
* sentence again. A retired benchmark is refused with `BENCHMARK_RETIRED`,
|
|
831
|
+
* because a rule pointed at one would report no number on every run and say
|
|
832
|
+
* nothing about why.
|
|
833
|
+
*/
|
|
834
|
+
setTrainingRuleBenchmark(id: string, params: TrainingRuleBenchmarkRequest): Promise<TrainingRuleMutationResponse>;
|
|
835
|
+
/**
|
|
836
|
+
* The workspace's agent options: which model it defaults to, the system
|
|
837
|
+
* prompts it judges and samples with, and the monthly cap on what its model
|
|
838
|
+
* calls may spend.
|
|
839
|
+
*/
|
|
840
|
+
getAgentSettings(): Promise<AgentSettings>;
|
|
841
|
+
/**
|
|
842
|
+
* Change them. An absent key leaves that setting exactly where it is; a
|
|
843
|
+
* present null returns it to the platform default. Those are three
|
|
844
|
+
* instructions, not two, so `{}` changes nothing.
|
|
845
|
+
*
|
|
846
|
+
* `eval_monthly_cap_cents` is pushed to the workspace's managed key, so it
|
|
847
|
+
* caps what the agent can spend even if a rule's own ceilings are higher.
|
|
848
|
+
*/
|
|
849
|
+
updateAgentSettings(params: AgentSettingsRequest): Promise<AgentSettings>;
|
|
850
|
+
}
|