runbios-sdk 0.2.17 → 0.2.18-dev.271

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,41 +1,54 @@
1
1
  import type { HttpClient } from '../client.js';
2
- import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopGrader, LoopGraderParams, LoopLabel, LoopLabelCount, LoopGradeResult, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopBuildRule, LoopBuildRuleParams, LoopBuildRuleUpdateParams, LoopStats, LoopJudge, LoopJudgeParams, LoopAgentCredential, LoopAgentStatus, LoopSampleRun, LoopSampleRunParams, LoopJudgeRun, LoopJudgeRunItems, LoopJudgeRunItemStatus, LoopJudgeWork, LoopJudgeVerdict, LoopJudgeVerdictResult, TrainingRule, TrainingRulePreflight, TrainingRulePreflightRequest, TrainingRuleCreateRequest, TrainingRuleUpdateRequest, TrainingRuleConsentRequest, TrainingRuleListParams, TrainingRuleListResponse, TrainingRuleResponse, TrainingRuleMutationResponse, TrainingRuleDeleteResponse, TrainingRun, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, JudgeAgreement, JudgeAgreementParams, Benchmark, BenchmarkCreateParams, BenchmarkHistoryParams, BenchmarkHistoryResponse, BenchmarkItemListParams, BenchmarkItemsResponse, BenchmarkListParams, BenchmarkListResponse, BenchmarkRetireParams, BenchmarkRun, TrainingRuleBenchmarkRequest, AgentSettings, AgentSettingsRequest, Pipeline, PipelineListResponse } from '../types.js';
2
+ import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopImportParams, LoopImportResult, LoopSignal, LoopSignalParams, LoopLabel, LoopLabelCount, LoopConfig, LoopConfigParams, LoopStats, LoopTracePipelinesResponse, LoopMetricsHistory, LoopMetricsHistoryParams, TrainingRuleDeleteResponse, TrainingRunListParams, TrainingRunListResponse, TrainingRunResponse, TrainingRunActionResponse, TrainingRunPromoteRequest, TrainingRunRejectRequest, TrainingRunRollbackRequest, TrainingRunCancelRequest, Evaluation, EvaluationItemListParams, EvaluationItemsResponse, Benchmark, BenchmarkCreateParams, BenchmarkHistoryParams, BenchmarkHistoryResponse, BenchmarkItemListParams, BenchmarkItemsResponse, BenchmarkListParams, BenchmarkListResponse, BenchmarkRetireParams, BenchmarkRun, Pipeline, PipelineListParams, PipelineListResponse, PipelineWriteRequest, PipelineMutationResponse, LoopModelsResponse, PipelineImportParams, PipelineBenchmarksResponse, PipelineBenchmarkAttachParams, PipelineBenchmarkMoveParams, VersionCompareResponse, VersionCompareParams } from '../types.js';
3
3
  /**
4
- * The Conscious Loop -- capture what your model was asked and answered, record
5
- * whether it was right, and turn those judgements into training data.
4
+ * The Conscious Loop -- pipelines that keep improving a model from the
5
+ * conversations you send them.
6
6
  *
7
- * Nothing is captured until you turn it on for a source, and you can turn it
8
- * off again at any time. Anything that looks like a credential or a personal
9
- * detail is removed before the record is written, never afterwards.
7
+ * A pipeline is one use case: a model, the tags whose samples train it, when
8
+ * to train, and how a new version is promoted. Every time it trains it makes
9
+ * an attempt; an attempt that beats the live model becomes the next version.
10
+ *
11
+ * Nothing is captured until a source is switched on, and you can switch it
12
+ * off again at any time. Creating a pipeline switches on the source named by
13
+ * its own tag (`own_tag`), stamping that tag, and deleting it switches that
14
+ * source off again. Anything that looks like a credential or a personal detail
15
+ * is removed before the record is written, never afterwards.
10
16
  *
11
17
  * Requires an API key carrying `loop:read` and/or `loop:write`. Those scopes are
12
18
  * deliberately absent from the Read Only preset, because what they return is
13
19
  * your raw prompts and completions rather than catalog metadata.
14
20
  *
21
+ * Offered on the development environment only for now. Anywhere else every
22
+ * method here, reads included, rejects with a `ComingSoonError` whose
23
+ * `code` is `LOOP_COMING_SOON` (HTTP 403): a deliberate product state, not an
24
+ * outage, so retrying cannot succeed there.
25
+ *
15
26
  * @example
16
27
  * ```ts
17
- * // 1. Decide that this source is recorded.
18
- * await client.loop.setConfig('my-agent', { enabled: true, retention_days: 30 });
28
+ * // 1. A pipeline: a model it can train, and its own tag.
29
+ * const { models } = await client.loop.listModels();
30
+ * const { pipeline } = await client.loop.createPipeline({
31
+ * name: 'Support replies',
32
+ * model: { id: models[0].id },
33
+ * });
19
34
  *
20
- * // 2. Send what your model was asked and what it answered. Works whoever
21
- * // served it -- us, another provider, or your own servers.
22
- * const { trace_id } = await client.loop.capture({
23
- * deployment_id: 'my-agent',
35
+ * // 2. Send samples from your app to the pipeline's own source: its create
36
+ * // switched that source on, and it tags everything it records.
37
+ * const sent = await client.loop.capture({
38
+ * deployment_id: pipeline.own_tag!,
24
39
  * model: 'gpt-4o',
25
40
  * messages: [{ role: 'user', content: 'what is our refund window' }],
26
41
  * completion: 'thirty days',
27
42
  * });
28
43
  *
29
- * // 3. Say whether it was right. A rewrite is the most valuable answer here:
30
- * // the model learns your version AND learns to avoid its own.
31
- * await client.loop.signal(trace_id, {
32
- * verdict: 'edited',
33
- * correction: 'Thirty days from delivery, no questions asked.',
34
- * });
44
+ * // 3. Say whether it was right, on the trace_id the capture answered with.
45
+ * // A correction is the most valuable answer.
46
+ * if (sent.trace_id) {
47
+ * await client.loop.correct(sent.trace_id, 'Thirty days from delivery, no questions asked.');
48
+ * }
35
49
  *
36
- * // 4. Turn the judgements into a training set and take the file.
37
- * const ds = await client.loop.createDataset({ name: 'support', method: 'sft' });
38
- * const jsonl = await client.loop.downloadDataset(ds.id);
50
+ * // 4. Once it has its minimum, it trains; versions appear on the pipeline.
51
+ * const p = await client.loop.getPipeline(pipeline.rule_id);
39
52
  * ```
40
53
  */
41
54
  export declare class Loop {
@@ -52,19 +65,25 @@ export declare class Loop {
52
65
  *
53
66
  * The returned `trace_id` is OURS. If you pass your own `request_id` it is
54
67
  * kept as an idempotency handle: sending the same one again returns the same
55
- * trace rather than storing a second copy, so a retry is safe.
68
+ * trace rather than storing a second copy, so a retry is safe. Send feedback
69
+ * on that `trace_id`.
56
70
  *
57
- * When the source is not enabled this resolves with `captured: false` and a
58
- * reason instead of throwing -- capture must never be the thing that breaks
59
- * your application.
71
+ * `deployment_id` is the source. A pipeline's `own_tag` is a source its
72
+ * create switched on, and it stamps that tag on everything it records;
73
+ * `labels` add tags of your own, so the same conversation can feed other
74
+ * pipelines too.
75
+ *
76
+ * When the source is not switched on this resolves with `captured: false`
77
+ * and a reason instead of throwing -- capture must never be the thing that
78
+ * breaks your application.
60
79
  */
61
80
  capture(params: LoopCaptureParams): Promise<LoopCaptureResult>;
62
81
  /**
63
82
  * Bring data you already have into the loop.
64
83
  *
65
- * This does NOT create a training set. It creates conversations, in the same
66
- * place captured ones live and subject to the same review, rules, judges and
67
- * labels. A row becomes trainable when something says it is good, never
84
+ * This does NOT create a training set. It creates conversations (samples),
85
+ * in the same place captured ones live and subject to the same feedback and
86
+ * tags. A row becomes trainable when something says it is good, never
68
87
  * because it arrived in a file -- which is the one guarantee that separates
69
88
  * a corpus from a pile.
70
89
  *
@@ -185,36 +204,15 @@ export declare class Loop {
185
204
  /** Every verdict on one conversation, oldest first. */
186
205
  listSignals(traceId: string): Promise<LoopSignal[]>;
187
206
  /**
188
- * Submit an alternative answer to a prompt already captured.
189
- *
190
- * This is what makes preference learning scale. A human rewrite is the best
191
- * signal there is and the least available -- somebody has to sit down and
192
- * write it. Sample the model several times for the same prompt, score the
193
- * samples, and a DPO pair falls out automatically: best becomes chosen,
194
- * worst becomes rejected. Point a stronger model at the prompt instead and
195
- * the same machinery does distillation.
196
- *
197
- * Score them, or they cannot pair: an unscored alternative says nothing about
198
- * which answer anybody prefers. `score_source` is required alongside a score,
199
- * because precedence between a verifier, a judge and a person is the whole
200
- * reason we record who judged.
201
- *
202
- * A human correction always outranks any score.
203
- *
204
- * @example
205
- * ```ts
206
- * for (const sample of await sampleMyModel(prompt, 4)) {
207
- * await client.loop.addCandidate(traceId, {
208
- * completion: sample.text,
209
- * score: await myVerifier(sample.text),
210
- * score_source: 'verifier',
211
- * });
212
- * }
213
- * ```
214
- */
215
- addCandidate(traceId: string, params: LoopCandidateParams): Promise<LoopCandidate>;
216
- /** Every alternative answer recorded for one prompt, oldest first. */
217
- listCandidates(traceId: string): Promise<LoopCandidate[]>;
207
+ * The pipelines one conversation feeds: every live pipeline whose tags
208
+ * select it, and whether each is switched on.
209
+ *
210
+ * Being listed means the pipeline would select it, not that it will be
211
+ * trained on -- a set still drops answers it cannot use and duplicates, and
212
+ * holds some back to judge by; `note` says so. An empty list is a real
213
+ * answer: the conversation trains nothing until it carries a pipeline's tag.
214
+ */
215
+ getTracePipelines(traceId: string): Promise<LoopTracePipelinesResponse>;
218
216
  /**
219
217
  * Label a conversation, so it can be selected later.
220
218
  *
@@ -268,139 +266,6 @@ export declare class Loop {
268
266
  * conversations in it that nobody had put there.
269
267
  */
270
268
  listLabels(): Promise<LoopLabelCount[]>;
271
- /**
272
- * Build a training set from the feedback recorded so far.
273
- *
274
- * The three methods need genuinely different things, so one set cannot be
275
- * reshaped into another afterwards:
276
- *
277
- * - `sft` -- answers you marked right, and answers you rewrote.
278
- * - `dpo` -- answers you REWROTE, so a better and a worse version of the same
279
- * reply exist. Nothing else produces a pair.
280
- * - `grpo` -- answers with a value or fact they can be checked against, or
281
- * a judge rubric, which scores answers the run has not written yet.
282
- * - `kto` -- anything carrying a yes or a no, INCLUDING a thumbs-down with
283
- * nothing written. That row trains nothing under the other three methods,
284
- * which is why this one exists: it is the feedback people actually give.
285
- *
286
- * `holdout_percent` holds back your most recent work rather than a random
287
- * slice, so the evaluation measures whether the model generalised instead of
288
- * memorised the same week.
289
- *
290
- * The result always reports `rejected_counts`: why rows were left out. A
291
- * small set with a reason is useful; a small set without one is just alarming.
292
- */
293
- createDataset(params: LoopDatasetCreateParams): Promise<LoopDataset>;
294
- /** List training sets, newest first. */
295
- /**
296
- * Stand up a rule that builds a set whenever enough new work exists.
297
- *
298
- * The manual path asks somebody to notice that enough conversations have
299
- * been reviewed, remember which filters describe the slice they want, and
300
- * press build — every time. A rule is that instruction, stored.
301
- *
302
- * `spec` is the same selection `createDataset` takes and is replayed
303
- * verbatim, so an automatic set is identical to a hand-made one.
304
- *
305
- * `min_new_rows` counts only work reviewed SINCE THE LAST BUILD. Counting
306
- * the whole corpus would fire the rule every interval forever, because a
307
- * total that has crossed a threshold stays across it. It is at least 100,
308
- * and 100 when absent: a set built from fewer is too small to learn from.
309
- */
310
- createBuildRule(params: LoopBuildRuleParams): Promise<LoopBuildRule>;
311
- /**
312
- * Every build rule, with what each one last did and why.
313
- *
314
- * `last_reason` is the field worth reading: a rule quiet because it is
315
- * waiting looks exactly like one quiet because it is broken.
316
- */
317
- listBuildRules(): Promise<LoopBuildRule[]>;
318
- /** Stop a standing build rule. Sets it already produced are untouched. */
319
- deleteBuildRule(id: string): Promise<{
320
- deleted: boolean;
321
- }>;
322
- listDatasets(params?: LoopDatasetListParams): Promise<LoopDataset[]>;
323
- /** Read one training set and its curation report. */
324
- getDataset(id: string): Promise<LoopDataset>;
325
- /**
326
- * Page through the rows a training set actually contains.
327
- *
328
- * Worth reading before you spend money on a run: each row carries the address
329
- * of the conversation it was built from.
330
- */
331
- listDatasetItems(id: string, params?: {
332
- limit?: number;
333
- offset?: number;
334
- }): Promise<LoopDatasetItem[]>;
335
- /**
336
- * Download a training set as JSONL -- one training row per line, the format
337
- * every trainer in this space reads.
338
- *
339
- * `split` defaults to the training rows; pass `holdout` for the slice held
340
- * back, or `all` for both.
341
- */
342
- downloadDataset(id: string, split?: 'train' | 'holdout' | 'all'): Promise<string>;
343
- /**
344
- * Delete a training set.
345
- *
346
- * The conversations it was built from are untouched -- a set is a selection,
347
- * and discarding the selection must not discard the evidence. A set a
348
- * training run was trained on is refused `409 DATASET_IN_USE` and kept, as
349
- * the record of what that run learned from. See
350
- * `LoopDatasetDeleteRefusalCode`.
351
- */
352
- deleteDataset(id: string): Promise<{
353
- deleted: boolean;
354
- dataset_id: string;
355
- }>;
356
- /**
357
- * Write a rule that scores answers without a person.
358
- *
359
- * Human review is the most trustworthy feedback and the least available. A
360
- * grader is written once and applied to every answer afterwards: did it
361
- * contain the required phrase, did it parse as the schema you asked for, is
362
- * the number within tolerance of the known answer, did it call the function
363
- * it should have.
364
- *
365
- * Every kind here is DETERMINISTIC -- no model call, no network. That is why
366
- * a verifier outranks a judge when they disagree: it cannot be flattered and
367
- * it cannot drift between runs.
368
- *
369
- * `weight` is the multiplier: a criterion that matters twice as much gets
370
- * twice the weight, and the combined score is the weighted mean over the
371
- * rules that actually applied.
372
- *
373
- * `matches_gold` needs no `expected`: it compares each answer to the gold
374
- * answer recorded on that same conversation, or the human correction when
375
- * there is no gold, and scores sampled alternatives against the same gold.
376
- * A conversation with neither is skipped, not failed, so one rule checks
377
- * every labelled question without punishing the unlabelled ones. An
378
- * optional `tolerance` widens the match when both sides are bare numbers.
379
- *
380
- * A caution worth knowing before you write a set: a rule made only of
381
- * `forbidden` phrases is satisfied VACUOUSLY by an answer that says nothing.
382
- * Pair it with a `required` phrase, or you are rewarding silence.
383
- */
384
- createGrader(params: LoopGraderParams): Promise<LoopGrader>;
385
- /** Every rule this workspace has written. */
386
- listGraders(): Promise<LoopGrader[]>;
387
- deleteGrader(id: string): Promise<{
388
- deleted: boolean;
389
- grader_id: string;
390
- }>;
391
- /**
392
- * Apply this workspace's rules to one captured answer AND to every
393
- * alternative sampled for it.
394
- *
395
- * Grading both in one pass is the point: scoring only the original gives a
396
- * verdict, while scoring the samples as well gives the preference pair. Sample
397
- * your model, call this, and you have DPO data with nobody reading anything.
398
- *
399
- * A run where no rule could apply writes NOTHING -- no verdict, no scores.
400
- * Recording a zero that no rule produced would poison curation with a
401
- * judgement nobody reached, so `applied: 0` is reported instead.
402
- */
403
- grade(traceId: string): Promise<LoopGradeResult>;
404
269
  /** Every source this workspace has configured. */
405
270
  listConfigs(): Promise<LoopConfig[]>;
406
271
  /**
@@ -420,299 +285,23 @@ export declare class Loop {
420
285
  */
421
286
  setConfig(source: string, params: LoopConfigParams): Promise<LoopConfig>;
422
287
  /**
423
- * How much there is, and how much of it each method could actually use.
288
+ * How much there is, and how many samples a pipeline could train on now
289
+ * (`ready.sft`: the ones marked good or corrected).
424
290
  *
425
- * The `ready` numbers are upper bounds: duplicate questions are folded into
426
- * one row while a set is built, so the finished count can be lower.
291
+ * `ready.sft` is an upper bound: duplicates, the held-back set and
292
+ * conversations too long to train on come out of it when an attempt trains.
427
293
  */
428
294
  stats(): Promise<LoopStats>;
429
- /** Write a rubric: what makes an answer good here, and what to score. */
430
- createJudge(params: LoopJudgeParams): Promise<LoopJudge>;
431
- listJudges(): Promise<LoopJudge[]>;
432
- /**
433
- * Rewrite a rubric. Runs already recorded keep the instructions they used:
434
- * a verdict has to keep meaning what it meant when it was given.
435
- */
436
- updateJudge(judgeId: string, params: LoopJudgeParams): Promise<LoopJudge>;
437
- /**
438
- * Retire a rubric. Its runs go with it; the verdicts they wrote stay. A score
439
- * is evidence about an answer, and it does not stop being true because the
440
- * rubric was retired.
441
- */
442
- deleteJudge(judgeId: string): Promise<void>;
443
- /**
444
- * Open a run: select the conversations now, and freeze the rubric onto them.
445
- *
446
- * The selection is fixed at this moment on purpose. A run whose selection is
447
- * a live query silently grows as conversations arrive, so "we scored the
448
- * refunds slice" becomes a claim about a set that no longer exists.
449
- */
450
- startRun(judgeId: string, limit?: number): Promise<LoopJudgeRun>;
451
- listRuns(judgeId?: string): Promise<LoopJudgeRun[]>;
452
- getRun(runId: string): Promise<LoopJudgeRun>;
453
- /**
454
- * Take the next conversations to score, each rendered into a ready-to-send
455
- * prompt. Nothing is marked taken, so a caller that dies half way loses
456
- * nothing: ask again and the same items come back.
457
- */
458
- takeWork(runId: string, limit?: number): Promise<LoopJudgeWork>;
459
- /**
460
- * What happened to each conversation in a run, and why.
461
- *
462
- * `takeWork` hands out what is still PENDING, so a finished run answers it
463
- * with an empty list. This answers with every item and the outcome on it:
464
- * `status`, `scored_at`, and `error` — the reason the caller gave for an
465
- * item it could not score, which is where a wrong model slug or a refused
466
- * key actually shows up. A run that ends "scored 0, failed 3" is otherwise a
467
- * number with no detail behind it.
468
- *
469
- * Pass `status: 'failed'` for the usual question. At most 500 items come
470
- * back at a time; when `has_more` is true, call again with the `next_offset`
471
- * from the reply.
472
- */
473
- listRunItems(runId: string, opts?: {
474
- status?: LoopJudgeRunItemStatus;
475
- limit?: number;
476
- offset?: number;
477
- }): Promise<LoopJudgeRunItems>;
478
- /**
479
- * Hand the scores back. A verdict for a conversation outside this run, or
480
- * scoring something the rubric never asked for, is refused and reported in
481
- * `rejected` rather than silently dropped.
482
- *
483
- * Pass `finish` to close the run in the same call once you have nothing left
484
- * to send. To close a run WITHOUT scores, call `stopRun`: an empty verdict
485
- * list with `finish` does the same thing and reads like a mistake.
486
- */
487
- postVerdicts(runId: string, verdicts: LoopJudgeVerdict[], finish?: boolean): Promise<LoopJudgeVerdictResult>;
488
- /**
489
- * Close a run that has not finished.
490
- *
491
- * A judge may have one open run of its own at a time, so an open run BLOCKS
492
- * the next one, and the runs that most need closing are the ones nobody can
493
- * wait out: a run the agent has parked on an empty balance or a refused key
494
- * stays open until the reason is fixed or somebody stops it.
495
- *
496
- * Nothing is deleted. Every verdict already recorded stays recorded, the
497
- * counters keep saying how much of the selection was covered, and the
498
- * conversations the run was holding are free for the next one. The run ends
499
- * as `stopped` rather than `done`, so an interrupted pass and a completed
500
- * one do not read alike.
501
- *
502
- * A run that finished on its own is not rewritten: stopping one answers 409.
503
- *
504
- * Returns the stopped run itself, like `startRun` and `getRun`, not the
505
- * `{run}` envelope the service sends. Every other single-run method in this
506
- * class unwraps, and `Loop.stop_run` in the Python SDK does too, so a
507
- * caller who wrote `(await loop.stopRun(id)).status` against one of its
508
- * siblings is right here as well.
509
- */
510
- stopRun(runId: string): Promise<LoopJudgeRun>;
511
295
  /**
512
- * Can an agent run here, is one running, and is it on for this workspace.
513
- * `available` is about the environment; `online` about the worker;
514
- * `credential` about this workspace.
515
- */
516
- agentStatus(): Promise<LoopAgentStatus>;
517
- /**
518
- * Turn the agent on: mints the workspace's managed serverless key and
519
- * resumes the automatic judges a previous turn-off paused. After this,
520
- * automatic judges and sample runs make model calls billed to the
521
- * workspace. Idempotent.
522
- *
523
- * A judge whose model the serving gateway will not route is NOT resumed:
524
- * making it automatic would buy a run that fails every conversation. It
525
- * stays paused, `judges_still_paused` counts those, and each one carries
526
- * `auto_pause_reason` saying so. Point it at a model that is served and
527
- * turn the agent on again.
528
- *
529
- * `monthly_spend_cap_cents` caps what that key may spend on model calls in
530
- * a calendar month. Omitted = no cap on a fresh key, and an existing cap is
531
- * left as it is; sent while the agent is already on, it moves the cap on
532
- * the existing key without minting a new one. When the cap is reached the
533
- * agent's model calls are refused until next month and its runs pause with
534
- * that reason.
296
+ * How the loop moved over time: every attempt as a point, oldest first --
297
+ * its attempt number, the version it became (null when it did not), its
298
+ * base model, its win rate, what it cost -- and each day's feedback counted
299
+ * by verdict. `rule_id` narrows the attempts to one pipeline; the feedback
300
+ * counts are the workspace's. `truncated` says older attempts exist beyond
301
+ * `limit`.
535
302
  */
536
- enableAgent(opts?: {
537
- monthly_spend_cap_cents?: number;
538
- }): Promise<LoopAgentCredential>;
539
- /**
540
- * Turn the agent off: revokes its key and pauses every automatic judge in
541
- * the workspace (`judges_paused`; each judge shows `auto_paused`). Turning
542
- * it back on resumes exactly those judges. Open runs stop where they are
543
- * and continue if it is turned back on. Nothing already scored or written
544
- * is removed.
545
- */
546
- disableAgent(): Promise<{
547
- revoked: boolean;
548
- judges_paused: number;
549
- message: string;
550
- }>;
551
- /**
552
- * Ask the agent to write `n` alternative answers to each conversation in a
553
- * slice with `model`, score each with `judge_id`, and store them as
554
- * candidates. This is how preference pairs are made without a person
555
- * writing each one: the curation pass pairs the best sample against the
556
- * worst wherever the gap is real.
557
- *
558
- * Cost: up to `n` calls to write plus `n` to judge, per conversation, at the
559
- * workspace's serverless rate; `selection.sample` caps the conversations
560
- * (max 200) and `n` is capped at 8. Opening a run turns the agent on if it
561
- * is off.
562
- */
563
- createSampleRun(params: LoopSampleRunParams): Promise<LoopSampleRun>;
564
- listSampleRuns(): Promise<LoopSampleRun[]>;
565
- getSampleRun(runId: string): Promise<LoopSampleRun>;
566
- /**
567
- * Price a training rule before anyone agrees to it. Writes nothing.
568
- *
569
- * Takes the create body without `accept_terms` and answers with the estimate
570
- * a member has to see first: the pinned model revision, the worst hourly
571
- * price each GPU ladder can reach, how many hours each ceiling buys, any
572
- * refusals that would stop a create, the API keys that cannot follow a
573
- * cutover because they lack `deployments:read`, and `terms_text` -- the
574
- * exact sentence to show, with real figures in it.
575
- *
576
- * Send the `terms_version` it returns back in `createTrainingRule`. Read the
577
- * figures out of this response rather than inventing ceilings of your own.
578
- *
579
- * `valid: false` with an EMPTY `refusals` list is not nothing: check
580
- * `unreachable`, which names every peer the platform could not reach. The
581
- * rule is savable in that state and would be paused, but the estimate around
582
- * it is not trustworthy -- both `worst_hourly_*` are `0`, and so are both
583
- * `max_*_hours` beside them -- so do not quote those figures to anyone.
584
- * `model_revision` is empty unless training-service actually pinned a
585
- * commit; test `unreachable.length > 0` to tell a degraded pass from a
586
- * refusal.
587
- *
588
- * A `TRAINING_NOT_ENABLED` refusal is neither: training is in beta and this
589
- * organisation does not hold the grant. It is in `refusals`, never in
590
- * `unreachable`, and nothing in the request can lift it.
591
- */
592
- preflightTrainingRule(params: TrainingRulePreflightRequest): Promise<TrainingRulePreflight>;
593
- /**
594
- * Create a training rule and record the consent that pays for it.
595
- *
596
- * ACCEPTING THE TERMS AUTHORISES SPENDING FROM YOUR WALLET WHILE YOU ARE NOT
597
- * PRESENT. From here on the platform may, on its own schedule and without
598
- * asking again, build a training set, run a training job on rented GPUs,
599
- * book a second deployment to compare against the one serving your traffic,
600
- * and pay for the model calls that judge the two. Those charges come out of
601
- * the wallet of the member whose credentials make this call, up to the
602
- * ceilings in `money`, and they keep recurring for as long as the rule is
603
- * enabled. If `promotion.auto_promote` is set, the platform will also
604
- * re-point your public handle at the new model with nobody reviewing it.
605
- *
606
- * `accept_terms` is therefore required, and this method refuses to send the
607
- * request without it rather than letting the server decide. Call
608
- * {@link preflightTrainingRule} first, show the person whose wallet pays the
609
- * `terms_text` and the figures it returns, get an explicit yes, and send the
610
- * `terms_version` they were shown. Do not invent ceilings or price caps on
611
- * their behalf.
612
- *
613
- * Refused `409 TRAINING_NOT_ENABLED`, with nothing saved, while training is
614
- * in beta and this organisation does not hold the grant: do not retry, ask
615
- * the platform team to enable it. See `TrainingRuleSaveRefusalCode`.
616
- *
617
- * @example
618
- * ```ts
619
- * const estimate = await client.loop.preflightTrainingRule(draft);
620
- * // show estimate.terms_text and the ceilings to the member, get a yes
621
- * const rule = await client.loop.createTrainingRule({
622
- * ...draft,
623
- * accept_terms: { terms_version: estimate.terms_version },
624
- * });
625
- * ```
626
- */
627
- createTrainingRule(params: TrainingRuleCreateRequest): Promise<TrainingRule>;
628
- /**
629
- * Every training rule in the workspace, with what each one last did and why.
630
- *
631
- * `last_reason` and `paused_reason` are the fields worth reading: a rule
632
- * quiet because it is waiting looks exactly like one quiet because its
633
- * consent went stale.
634
- */
635
- listTrainingRules(params?: TrainingRuleListParams): Promise<TrainingRuleListResponse>;
636
- /**
637
- * Read one rule with its recent runs, what its runs this month have cost at
638
- * most, and how far its judge agrees with your own reviewers.
639
- *
640
- * Returned whole rather than unwrapped to the rule: `month_spent_cents` is
641
- * the number that says whether the monthly ceiling is about to stop it. It
642
- * is an upper bound, not an exact spend: see `Pipeline.month_spent_cents`.
643
- */
644
- getTrainingRule(id: string): Promise<TrainingRuleResponse>;
645
- /**
646
- * Edit or pause a rule. Absent means unchanged; a null clears the fields
647
- * that can be cleared.
648
- *
649
- * A money-bearing change -- a ceiling, a model, a GPU ladder, the promotion
650
- * policy, or changing `explore_recipes` in EITHER direction -- bumps
651
- * `revision`, clears the recorded consent and STOPS the rule firing until
652
- * someone accepts the new amounts. Turning recipe variants OFF does this
653
- * too: the pipeline then makes no version at all, variant or not, until the
654
- * terms are accepted again. The reply says so in
655
- * `consent_required`, and carries a fresh `preflight` with the new figures.
656
- * Pass `expected_revision` to be refused with `REVISION_MISMATCH` rather
657
- * than overwrite an edit somebody else made in the meantime.
658
- *
659
- * Switching a rule on, switching it into preference training or carrying
660
- * `accept_terms` re-runs the preflight, and can be refused
661
- * `409 TRAINING_NOT_ENABLED` like a create. A rule paused
662
- * `training_not_enabled` is started again by a save of it once the
663
- * organisation has been enabled.
664
- */
665
- updateTrainingRule(id: string, params: TrainingRuleUpdateRequest): Promise<TrainingRuleMutationResponse>;
666
- /**
667
- * Accept the rule's current terms, so it may fire again.
668
- *
669
- * ACCEPTING THE TERMS AUTHORISES SPENDING FROM YOUR WALLET WHILE YOU ARE NOT
670
- * PRESENT, on the amounts as they stand right now. This is the same
671
- * authorisation {@link createTrainingRule} records, given again because a
672
- * money-bearing edit cleared the old one: training, a candidate deployment
673
- * and the judge's model calls are charged to the wallet of the member making
674
- * this call, up to the rule's ceilings, every time it fires.
675
- *
676
- * Show the member the current `terms_text` from a fresh
677
- * {@link preflightTrainingRule} or from the `preflight` on the update reply,
678
- * and send the `revision` those figures belong to. A stale revision is
679
- * refused with `409`, which is the point: it means the amounts moved again
680
- * after they were read. `409 TRAINING_NOT_ENABLED` means training is in beta
681
- * and this organisation does not hold the grant.
682
- */
683
- consentTrainingRule(id: string, params: TrainingRuleConsentRequest): Promise<TrainingRule>;
684
- /**
685
- * Fire a rule now, without waiting for its cadence.
686
- *
687
- * Bypasses the schedule and `min_new_rows` only. The row floors, the money
688
- * ceilings, the consent, the version limit and the monthly limit all still
689
- * apply, so this can answer `409 CONSENT_REQUIRED`, `409 RULE_PAUSED`,
690
- * `409 RUN_ACTIVE`, `409 VERSION_LIMIT_REACHED` (the pipeline has made
691
- * `max_versions` versions), `409 MONTHLY_LIMIT_REACHED` (the most this
692
- * month's runs can have cost plus the most one run may cost would pass the
693
- * monthly limit; the error
694
- * body carries `month_spent_cents`, `monthly_ceiling_cents`, `run_max_cents`
695
- * and `resumes_at`), `409 AGENT_OFF` (the workspace's Conscious Loop agent
696
- * is not turned on, so a trained model could not be compared) or
697
- * `422 NOT_ENOUGH_ROWS` with the counts it needed. See
698
- * `TrainingRuleRunRefusalCode`.
699
- */
700
- runTrainingRule(id: string): Promise<TrainingRun>;
701
- /**
702
- * Delete a rule. An active run is cancelled; runs that already finished, and
703
- * anything already promoted, are kept.
704
- */
705
- deleteTrainingRule(id: string): Promise<TrainingRuleDeleteResponse>;
706
- /**
707
- * Edit or pause a build rule -- the standing instruction that assembles the
708
- * training set a training rule then trains on.
709
- *
710
- * Changing `spec.deployment_id` while an enabled training rule owns this
711
- * build rule is refused with `409`: it would silently retrain the next model
712
- * on a different source's conversations.
713
- */
714
- updateBuildRule(id: string, params: LoopBuildRuleUpdateParams): Promise<LoopBuildRule>;
715
- /** Firings, newest first. Filter by rule or by the state they are sitting in. */
303
+ getMetricsHistory(params?: LoopMetricsHistoryParams): Promise<LoopMetricsHistory>;
304
+ /** Attempts, newest first. Filter by pipeline (`rule_id`) or by the state they are sitting in. */
716
305
  listTrainingRuns(params?: TrainingRunListParams): Promise<TrainingRunListResponse>;
717
306
  /**
718
307
  * One run with its timeline, the addresses of everything it created, and
@@ -751,24 +340,19 @@ export declare class Loop {
751
340
  */
752
341
  cancelTrainingRun(id: string, params?: TrainingRunCancelRequest): Promise<TrainingRunActionResponse>;
753
342
  /**
754
- * Every training rule in the workspace, seen as the series of versions it
755
- * produced: which exist, which one serves (`champion_version`), how each did
756
- * against the champion of its day and on the standing benchmark, the run in
757
- * flight and which version it will be, and what the pipeline is waiting for.
758
- *
759
- * Read-only. Every decision stays on the route that owns it -- promote,
760
- * reject and roll back on the run, the version limit and `explore_recipes`
761
- * on the rule.
343
+ * Every pipeline in the workspace: its versions (the attempts that became
344
+ * the live model), every attempt, which version is live, the attempt in
345
+ * flight, and what the pipeline is waiting for.
762
346
  *
763
347
  * RETURNED WITH ITS ENVELOPE, because the list is not always all of them:
764
- * the service returns the newest 100, `total` is every pipeline in the
765
- * workspace and `truncated` is true when more exist than were returned. A
766
- * caller that shows `pipelines` alone presents a short list as the whole of
767
- * it. The rest are reached through `listTrainingRules`, which pages.
348
+ * the service returns a page of 100, newest first, `total` is every
349
+ * pipeline in the workspace and `truncated` is true when more exist after
350
+ * this page. A caller that shows `pipelines` alone presents a short list as
351
+ * the whole of it; the next page is `{ offset: offset + pipelines.length }`.
768
352
  */
769
- listPipelines(): Promise<PipelineListResponse>;
353
+ listPipelines(params?: PipelineListParams): Promise<PipelineListResponse>;
770
354
  /**
771
- * One pipeline, by its training rule's id.
355
+ * One pipeline, by its id.
772
356
  *
773
357
  * `status` says whether it is the platform working (`running`), the member
774
358
  * who has to act (`needs_review`, `needs_funds`), or nothing at all until
@@ -778,9 +362,71 @@ export declare class Loop {
778
362
  * at most, not an exact spend.
779
363
  */
780
364
  getPipeline(id: string): Promise<Pipeline>;
365
+ /**
366
+ * The base models a pipeline can train: only those the engine trains with
367
+ * LoRA SFT and the platform can serve for the comparison, recommended first.
368
+ * The one list {@link createPipeline} accepts a `model.id` from, with the
369
+ * `floors` a new pipeline is held to: the samples its first attempt needs and
370
+ * the smallest `train_when.min_samples`.
371
+ */
372
+ listModels(): Promise<LoopModelsResponse>;
373
+ /**
374
+ * Create a pipeline -- one use case.
375
+ *
376
+ * SAVING IT AUTHORISES SPENDING FROM YOUR WALLET WHILE YOU ARE NOT PRESENT:
377
+ * whenever its data has grown enough (and at its schedule, when it has one),
378
+ * the platform trains a new version on GPUs it chooses, compares it with what
379
+ * serves your traffic, and promotes it by the pipeline's promotion policy.
380
+ * What is charged is the GPU time the runs actually use; the platform's own
381
+ * per-run and monthly limits stop a run that goes wrong.
382
+ */
383
+ createPipeline(params: PipelineWriteRequest): Promise<PipelineMutationResponse>;
384
+ /**
385
+ * Edit a pipeline. Only the fields sent change; `max_versions: null` removes
386
+ * the limit and `challenger_model: null` removes the challenger. `promotion`
387
+ * is how the promotion policy is set. Saving is the authorization, as on a
388
+ * create.
389
+ */
390
+ updatePipeline(id: string, params: PipelineWriteRequest): Promise<PipelineMutationResponse>;
391
+ /**
392
+ * Delete a pipeline: an attempt still running is cancelled; the versions it
393
+ * made, what serves your app, and its samples and tags are kept. Recording
394
+ * to its own tag stops.
395
+ */
396
+ deletePipeline(id: string): Promise<TrainingRuleDeleteResponse>;
397
+ /**
398
+ * Train now: skips the schedule, never the minimum. A pipeline short of its
399
+ * samples is refused 422 `NOT_ENOUGH_SAMPLES` with `have` and `need` -- the
400
+ * numbers `minimums` shows. 503 `AGENT_COULD_NOT_START` means the key the
401
+ * attempt would be scored with could not be renewed just then: nothing was
402
+ * started, so press train now again in a minute. Every refusal is a
403
+ * `TrainingRuleRunRefusalCode`.
404
+ */
405
+ runPipeline(id: string): Promise<PipelineMutationResponse>;
406
+ /**
407
+ * {@link importRows}, with every row given the pipeline's own tag. `source`
408
+ * defaults to that tag; the rows, the answer and the retry rules are the
409
+ * import door's own. `default_feedback: 'good'` records every answered row
410
+ * that brought no verdict of its own as a good example to learn from.
411
+ */
412
+ importIntoPipeline(id: string, params: PipelineImportParams): Promise<LoopImportResult>;
413
+ /** The pipeline's benchmarks in priority order (1 is the primary), the ones it stopped using, and its promotion policy. */
414
+ listPipelineBenchmarks(id: string): Promise<PipelineBenchmarksResponse>;
415
+ /** Add a benchmark to a pipeline, last unless `priority` says where. Every version from now on is replayed on it. */
416
+ attachPipelineBenchmark(id: string, params: PipelineBenchmarkAttachParams): Promise<PipelineBenchmarksResponse>;
417
+ /** Move a benchmark on a pipeline's list; priority 1 makes it the primary. */
418
+ movePipelineBenchmark(id: string, benchmarkId: string, params: PipelineBenchmarkMoveParams): Promise<PipelineBenchmarksResponse>;
419
+ /** Stop using a benchmark. Every score it produced is kept. */
420
+ removePipelineBenchmark(id: string, benchmarkId: string): Promise<PipelineBenchmarksResponse>;
421
+ /**
422
+ * Every version's scores side by side: its head-to-head and every benchmark,
423
+ * `not_measured` where a benchmark was added after it. `a` and `b` are
424
+ * VERSION numbers; with both, `differences` says b minus a per benchmark.
425
+ */
426
+ compareVersions(id: string, params?: VersionCompareParams): Promise<VersionCompareResponse>;
781
427
  /**
782
428
  * The comparison behind a verdict: both models on the same held-out rows,
783
- * with identical decoding, judge and grader names resolved.
429
+ * with identical decoding, scored by the same judge.
784
430
  *
785
431
  * Read `warnings` before you read `win_rate`. A win rate over a holdout too
786
432
  * small to mean anything, or one measured by a judge that disagrees with
@@ -794,20 +440,12 @@ export declare class Loop {
794
440
  * The paired conversations behind the numbers: one prompt, both answers, the
795
441
  * scores each earned, and which won.
796
442
  *
797
- * Filter by `winner` to read the losses first, which is where a verdict is
798
- * actually checked. `limit` is capped at 100 by the service.
443
+ * Filter by `winner` -- `candidate` (the attempt won), `serving` (the version
444
+ * serving today won; the item itself says `incumbent`) or `tie` -- and read
445
+ * `serving` first: the pairs the attempt lost are where a verdict is actually
446
+ * checked. `limit` is 50 when absent and at most 200.
799
447
  */
800
448
  listEvaluationItems(id: string, params?: EvaluationItemListParams): Promise<EvaluationItemsResponse>;
801
- /**
802
- * How far a judge agrees with your own reviewers, over the conversations
803
- * both have scored.
804
- *
805
- * This is a gate, not a badge: a rule's `min_judge_agreement` refuses to
806
- * promote on the word of a judge that does not agree with the people whose
807
- * product it is. `enough_pairs` is the field to read first -- "not enough
808
- * reviewer overlap yet" is an answer, and 100% of two pairs is not.
809
- */
810
- getJudgeAgreement(judgeId: string, params?: JudgeAgreementParams): Promise<JudgeAgreement>;
811
449
  /**
812
450
  * Pin a fixed set of conversations, with a judge frozen beside them, and
813
451
  * measure every future run against it.
@@ -823,9 +461,10 @@ export declare class Loop {
823
461
  * is refused rather than skipped, because pinning 47 of the 50 you chose is
824
462
  * the set being wrong from the first day and you would never find out.
825
463
  *
826
- * It raises no amount you have already agreed to. The replay's calls come
827
- * out of the rule's existing `eval_ceiling_cents`, and
828
- * `per_run_ceiling_cents` can only lower what is spent inside that.
464
+ * The platform freezes the judge and the model it scores on (the ones every
465
+ * pipeline comparison uses), gives both models the comparison's own answer
466
+ * length, and sets what one replay may spend, inside the attempt's own
467
+ * scoring limit; none of that is sent.
829
468
  */
830
469
  createBenchmark(params: BenchmarkCreateParams): Promise<Benchmark>;
831
470
  /** Your benchmarks, newest first. Filter by `status` to hide retired ones. */
@@ -840,7 +479,7 @@ export declare class Loop {
840
479
  */
841
480
  getBenchmark(id: string): Promise<Benchmark>;
842
481
  /**
843
- * The pinned conversations, paged. `limit` is capped at 100 by the service.
482
+ * The pinned conversations, paged. `limit` is 50 when absent and at most 200.
844
483
  *
845
484
  * `source_trace_id` and `source_trace_url` come back null once the
846
485
  * conversation a row was copied from has been deleted. The row itself stays
@@ -859,7 +498,7 @@ export declare class Loop {
859
498
  */
860
499
  getBenchmarkHistory(id: string, params?: BenchmarkHistoryParams): Promise<BenchmarkHistoryResponse>;
861
500
  /**
862
- * Stop replaying a benchmark, and detach it from every rule that names it.
501
+ * Stop replaying a benchmark, and detach it from every pipeline using it.
863
502
  *
864
503
  * Nothing measured is removed. There is no delete and no update on purpose:
865
504
  * the series of numbers measured against a set is what a benchmark is for,
@@ -883,36 +522,4 @@ export declare class Loop {
883
522
  * different set, which is the exact defect a standing benchmark removes.
884
523
  */
885
524
  getBenchmarkRun(id: string): Promise<BenchmarkRun>;
886
- /**
887
- * Attach a benchmark to a rule, or send `{ benchmark_id: null }` to detach
888
- * it.
889
- *
890
- * Its own route rather than a field on the rule body, because it is a
891
- * decision to replay a fixed set on every future run of this rule for as
892
- * long as it stands, and its refusals -- retired, or belonging to another
893
- * workspace -- are about the benchmark rather than about the rule.
894
- *
895
- * It does not invalidate consent and the reply says so: attaching raises
896
- * neither the amount set aside for judge calls nor the amount set aside for
897
- * keeping the new model available, so nobody is asked to read the same
898
- * sentence again. A retired benchmark is refused with `BENCHMARK_RETIRED`,
899
- * because a rule pointed at one would report no number on every run and say
900
- * nothing about why.
901
- */
902
- setTrainingRuleBenchmark(id: string, params: TrainingRuleBenchmarkRequest): Promise<TrainingRuleMutationResponse>;
903
- /**
904
- * The workspace's agent options: which model it defaults to, the system
905
- * prompts it judges and samples with, and the monthly cap on what its model
906
- * calls may spend.
907
- */
908
- getAgentSettings(): Promise<AgentSettings>;
909
- /**
910
- * Change them. An absent key leaves that setting exactly where it is; a
911
- * present null returns it to the platform default. Those are three
912
- * instructions, not two, so `{}` changes nothing.
913
- *
914
- * `eval_monthly_cap_cents` is pushed to the workspace's managed key, so it
915
- * caps what the agent can spend even if a rule's own ceilings are higher.
916
- */
917
- updateAgentSettings(params: AgentSettingsRequest): Promise<AgentSettings>;
918
525
  }