runbios-sdk 0.2.1-dev.105 → 0.2.1-dev.106

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/index.d.ts CHANGED
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
35
35
  * SDK version. Sent as part of the User-Agent header.
36
36
  * Must match package.json "version" -- enforced by a contract test.
37
37
  */
38
- export declare const VERSION = "0.2.1-dev.105";
38
+ export declare const VERSION = "0.2.1-dev.106";
39
39
  export declare class RunBiOS {
40
40
  /** Search models, fetch configs, check adapter compatibility. */
41
41
  readonly models: Models;
package/dist/index.js CHANGED
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
35
35
  * SDK version. Sent as part of the User-Agent header.
36
36
  * Must match package.json "version" -- enforced by a contract test.
37
37
  */
38
- export const VERSION = '0.2.1-dev.105';
38
+ export const VERSION = '0.2.1-dev.106';
39
39
  export class RunBiOS {
40
40
  /** Search models, fetch configs, check adapter compatibility. */
41
41
  models;
@@ -1,5 +1,5 @@
1
1
  import type { HttpClient } from '../client.js';
2
- import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats } from '../types.js';
2
+ import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCaptureParams, LoopCaptureResult, LoopSignal, LoopSignalParams, LoopCandidate, LoopCandidateParams, LoopDataset, LoopDatasetListParams, LoopDatasetItem, LoopDatasetCreateParams, LoopConfig, LoopConfigParams, LoopStats } from '../types.js';
3
3
  /**
4
4
  * The Conscious Loop -- capture what your model was asked and answered, record
5
5
  * whether it was right, and turn those judgements into training data.
@@ -18,7 +18,7 @@ import type { LoopTrace, LoopTraceListResponse, LoopTraceListParams, LoopCapture
18
18
  * await client.loop.setConfig('my-agent', { enabled: true, retention_days: 30 });
19
19
  *
20
20
  * // 2. Send what your model was asked and what it answered. Works whoever
21
- * // served it -- us, OpenAI, your own vLLM.
21
+ * // served it -- us, another provider, or your own servers.
22
22
  * const { trace_id } = await client.loop.capture({
23
23
  * deployment_id: 'my-agent',
24
24
  * model: 'gpt-4o',
@@ -46,7 +46,7 @@ export declare class Loop {
46
46
  * Send one exchange to be captured.
47
47
  *
48
48
  * Works regardless of who served the model: ours, OpenAI, Fireworks, your own
49
- * vLLM, an agent framework. Pass `tool_calls` and `tools` when the turn
49
+ * own servers, an agent framework. Pass `tool_calls` and `tools` when the turn
50
50
  * invoked a function -- without them a tool-using exchange trains the model to
51
51
  * answer in prose where it should have called something.
52
52
  *
@@ -90,6 +90,37 @@ export declare class Loop {
90
90
  signal(traceId: string, params: LoopSignalParams): Promise<LoopSignal>;
91
91
  /** Every verdict on one conversation, oldest first. */
92
92
  listSignals(traceId: string): Promise<LoopSignal[]>;
93
+ /**
94
+ * Submit an alternative answer to a prompt already captured.
95
+ *
96
+ * This is what makes preference learning scale. A human rewrite is the best
97
+ * signal there is and the least available -- somebody has to sit down and
98
+ * write it. Sample the model several times for the same prompt, score the
99
+ * samples, and a DPO pair falls out automatically: best becomes chosen,
100
+ * worst becomes rejected. Point a stronger model at the prompt instead and
101
+ * the same machinery does distillation.
102
+ *
103
+ * Score them, or they cannot pair: an unscored alternative says nothing about
104
+ * which answer anybody prefers. `score_source` is required alongside a score,
105
+ * because precedence between a verifier, a judge and a person is the whole
106
+ * reason we record who judged.
107
+ *
108
+ * A human correction always outranks any score.
109
+ *
110
+ * @example
111
+ * ```ts
112
+ * for (const sample of await sampleMyModel(prompt, 4)) {
113
+ * await client.loop.addCandidate(traceId, {
114
+ * completion: sample.text,
115
+ * score: await myVerifier(sample.text),
116
+ * score_source: 'verifier',
117
+ * });
118
+ * }
119
+ * ```
120
+ */
121
+ addCandidate(traceId: string, params: LoopCandidateParams): Promise<LoopCandidate>;
122
+ /** Every alternative answer recorded for one prompt, oldest first. */
123
+ listCandidates(traceId: string): Promise<LoopCandidate[]>;
93
124
  /**
94
125
  * Build a training set from the feedback recorded so far.
95
126
  *
@@ -16,7 +16,7 @@
16
16
  * await client.loop.setConfig('my-agent', { enabled: true, retention_days: 30 });
17
17
  *
18
18
  * // 2. Send what your model was asked and what it answered. Works whoever
19
- * // served it -- us, OpenAI, your own vLLM.
19
+ * // served it -- us, another provider, or your own servers.
20
20
  * const { trace_id } = await client.loop.capture({
21
21
  * deployment_id: 'my-agent',
22
22
  * model: 'gpt-4o',
@@ -47,7 +47,7 @@ export class Loop {
47
47
  * Send one exchange to be captured.
48
48
  *
49
49
  * Works regardless of who served the model: ours, OpenAI, Fireworks, your own
50
- * vLLM, an agent framework. Pass `tool_calls` and `tools` when the turn
50
+ * own servers, an agent framework. Pass `tool_calls` and `tools` when the turn
51
51
  * invoked a function -- without them a tool-using exchange trains the model to
52
52
  * answer in prose where it should have called something.
53
53
  *
@@ -121,6 +121,44 @@ export class Loop {
121
121
  const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`);
122
122
  return res.signals;
123
123
  }
124
+ // ── alternative answers ───────────────────────────────────────────────
125
+ /**
126
+ * Submit an alternative answer to a prompt already captured.
127
+ *
128
+ * This is what makes preference learning scale. A human rewrite is the best
129
+ * signal there is and the least available -- somebody has to sit down and
130
+ * write it. Sample the model several times for the same prompt, score the
131
+ * samples, and a DPO pair falls out automatically: best becomes chosen,
132
+ * worst becomes rejected. Point a stronger model at the prompt instead and
133
+ * the same machinery does distillation.
134
+ *
135
+ * Score them, or they cannot pair: an unscored alternative says nothing about
136
+ * which answer anybody prefers. `score_source` is required alongside a score,
137
+ * because precedence between a verifier, a judge and a person is the whole
138
+ * reason we record who judged.
139
+ *
140
+ * A human correction always outranks any score.
141
+ *
142
+ * @example
143
+ * ```ts
144
+ * for (const sample of await sampleMyModel(prompt, 4)) {
145
+ * await client.loop.addCandidate(traceId, {
146
+ * completion: sample.text,
147
+ * score: await myVerifier(sample.text),
148
+ * score_source: 'verifier',
149
+ * });
150
+ * }
151
+ * ```
152
+ */
153
+ async addCandidate(traceId, params) {
154
+ const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/candidates`, params);
155
+ return res.candidate;
156
+ }
157
+ /** Every alternative answer recorded for one prompt, oldest first. */
158
+ async listCandidates(traceId) {
159
+ const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/candidates`);
160
+ return res.candidates;
161
+ }
124
162
  // ── training sets ─────────────────────────────────────────────────────
125
163
  /**
126
164
  * Build a training set from the feedback recorded so far.
package/dist/types.d.ts CHANGED
@@ -2140,3 +2140,30 @@ export interface LoopStats {
2140
2140
  /** Upper bounds: deduplication runs when a set is built. */
2141
2141
  ready: Record<LoopMethod, number>;
2142
2142
  }
2143
+ export interface LoopCandidateParams {
2144
+ completion?: string;
2145
+ /** An alternative that is itself a tool call. */
2146
+ tool_calls?: LoopToolCall[];
2147
+ /** Which model produced this alternative. The teacher, when distilling. */
2148
+ model?: string;
2149
+ model_version?: string;
2150
+ /** Unscored alternatives cannot pair -- a missing score is not a low one. */
2151
+ score?: number;
2152
+ /** Required alongside a score: human | verifier | judge | behavioural. */
2153
+ score_source?: LoopSource;
2154
+ reason?: string;
2155
+ metadata?: Record<string, unknown>;
2156
+ }
2157
+ export interface LoopCandidate {
2158
+ id: string;
2159
+ trace_id: string;
2160
+ model: string | null;
2161
+ model_version: string | null;
2162
+ completion: string;
2163
+ tool_calls?: LoopToolCall[];
2164
+ score: number | null;
2165
+ score_source: LoopSource | null;
2166
+ reason: string | null;
2167
+ metadata: Record<string, unknown>;
2168
+ created_at: string;
2169
+ }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "runbios-sdk",
3
- "version": "0.2.1-dev.105",
3
+ "version": "0.2.1-dev.106",
4
4
  "description": "Official TypeScript SDK for the Run BiOS training and deployment platform API",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",