runbios-sdk 0.2.1-dev.107 → 0.2.1-dev.108
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +1 -1
- package/dist/index.js +1 -1
- package/dist/resources/loop.d.ts +36 -0
- package/dist/resources/loop.js +44 -0
- package/dist/types.d.ts +9 -1
- package/package.json +1 -1
package/dist/index.d.ts
CHANGED
|
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
|
|
|
35
35
|
* SDK version. Sent as part of the User-Agent header.
|
|
36
36
|
* Must match package.json "version" -- enforced by a contract test.
|
|
37
37
|
*/
|
|
38
|
-
export declare const VERSION = "0.2.1-dev.
|
|
38
|
+
export declare const VERSION = "0.2.1-dev.108";
|
|
39
39
|
export declare class RunBiOS {
|
|
40
40
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
41
41
|
readonly models: Models;
|
package/dist/index.js
CHANGED
|
@@ -35,7 +35,7 @@ import { Loop } from './resources/loop.js';
|
|
|
35
35
|
* SDK version. Sent as part of the User-Agent header.
|
|
36
36
|
* Must match package.json "version" -- enforced by a contract test.
|
|
37
37
|
*/
|
|
38
|
-
export const VERSION = '0.2.1-dev.
|
|
38
|
+
export const VERSION = '0.2.1-dev.108';
|
|
39
39
|
export class RunBiOS {
|
|
40
40
|
/** Search models, fetch configs, check adapter compatibility. */
|
|
41
41
|
models;
|
package/dist/resources/loop.d.ts
CHANGED
|
@@ -88,6 +88,42 @@ export declare class Loop {
|
|
|
88
88
|
* behavioural hint.
|
|
89
89
|
*/
|
|
90
90
|
signal(traceId: string, params: LoopSignalParams): Promise<LoopSignal>;
|
|
91
|
+
/**
|
|
92
|
+
* Thumbs up: this answer was good. Shorthand for `signal(id, { verdict:
|
|
93
|
+
* 'accepted' })` — there is one stored vocabulary behind both, so a thumbs-up
|
|
94
|
+
* and an "accepted" can never disagree in your corpus.
|
|
95
|
+
*/
|
|
96
|
+
upvote(traceId: string, reason?: string): Promise<LoopSignal>;
|
|
97
|
+
/**
|
|
98
|
+
* Thumbs down: this answer was bad, and you have nothing better to offer.
|
|
99
|
+
*
|
|
100
|
+
* On its own this REMOVES the answer from training rather than teaching
|
|
101
|
+
* anything — there is no better version to learn. If you know what it should
|
|
102
|
+
* have said, `correct()` is worth far more.
|
|
103
|
+
*/
|
|
104
|
+
downvote(traceId: string, reason?: string): Promise<LoopSignal>;
|
|
105
|
+
/**
|
|
106
|
+
* The model was wrong; here is what it should have said.
|
|
107
|
+
*
|
|
108
|
+
* The most valuable feedback there is, because it produces BOTH halves of a
|
|
109
|
+
* preference pair from one action: the model learns your answer and learns to
|
|
110
|
+
* avoid its own.
|
|
111
|
+
*/
|
|
112
|
+
correct(traceId: string, answer: string, reason?: string): Promise<LoopSignal>;
|
|
113
|
+
/**
|
|
114
|
+
* Record the GOLD answer for this question — the reference, regardless of
|
|
115
|
+
* what the model happened to say.
|
|
116
|
+
*
|
|
117
|
+
* Different from `correct()` in a way that matters: a correction asserts the
|
|
118
|
+
* model was wrong, so the original becomes the rejected half of a pair. A gold
|
|
119
|
+
* answer asserts nothing about the model, so if it MATCHES what was said, no
|
|
120
|
+
* preference pair is invented — but SFT still learns it.
|
|
121
|
+
*
|
|
122
|
+
* It is the most reusable thing you can record: it trains SFT, forms a DPO
|
|
123
|
+
* pair when it differs from the answer, and stands in as the GRPO reference
|
|
124
|
+
* when you have not supplied a separate ground truth.
|
|
125
|
+
*/
|
|
126
|
+
gold(traceId: string, answer: string, reason?: string): Promise<LoopSignal>;
|
|
91
127
|
/** Every verdict on one conversation, oldest first. */
|
|
92
128
|
listSignals(traceId: string): Promise<LoopSignal[]>;
|
|
93
129
|
/**
|
package/dist/resources/loop.js
CHANGED
|
@@ -116,6 +116,50 @@ export class Loop {
|
|
|
116
116
|
const res = await this._http.fetchPost(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`, params);
|
|
117
117
|
return res.signal;
|
|
118
118
|
}
|
|
119
|
+
/**
|
|
120
|
+
* Thumbs up: this answer was good. Shorthand for `signal(id, { verdict:
|
|
121
|
+
* 'accepted' })` — there is one stored vocabulary behind both, so a thumbs-up
|
|
122
|
+
* and an "accepted" can never disagree in your corpus.
|
|
123
|
+
*/
|
|
124
|
+
async upvote(traceId, reason) {
|
|
125
|
+
return this.signal(traceId, { verdict: 'accepted', reason });
|
|
126
|
+
}
|
|
127
|
+
/**
|
|
128
|
+
* Thumbs down: this answer was bad, and you have nothing better to offer.
|
|
129
|
+
*
|
|
130
|
+
* On its own this REMOVES the answer from training rather than teaching
|
|
131
|
+
* anything — there is no better version to learn. If you know what it should
|
|
132
|
+
* have said, `correct()` is worth far more.
|
|
133
|
+
*/
|
|
134
|
+
async downvote(traceId, reason) {
|
|
135
|
+
return this.signal(traceId, { verdict: 'rejected', reason });
|
|
136
|
+
}
|
|
137
|
+
/**
|
|
138
|
+
* The model was wrong; here is what it should have said.
|
|
139
|
+
*
|
|
140
|
+
* The most valuable feedback there is, because it produces BOTH halves of a
|
|
141
|
+
* preference pair from one action: the model learns your answer and learns to
|
|
142
|
+
* avoid its own.
|
|
143
|
+
*/
|
|
144
|
+
async correct(traceId, answer, reason) {
|
|
145
|
+
return this.signal(traceId, { verdict: 'edited', correction: answer, reason });
|
|
146
|
+
}
|
|
147
|
+
/**
|
|
148
|
+
* Record the GOLD answer for this question — the reference, regardless of
|
|
149
|
+
* what the model happened to say.
|
|
150
|
+
*
|
|
151
|
+
* Different from `correct()` in a way that matters: a correction asserts the
|
|
152
|
+
* model was wrong, so the original becomes the rejected half of a pair. A gold
|
|
153
|
+
* answer asserts nothing about the model, so if it MATCHES what was said, no
|
|
154
|
+
* preference pair is invented — but SFT still learns it.
|
|
155
|
+
*
|
|
156
|
+
* It is the most reusable thing you can record: it trains SFT, forms a DPO
|
|
157
|
+
* pair when it differs from the answer, and stands in as the GRPO reference
|
|
158
|
+
* when you have not supplied a separate ground truth.
|
|
159
|
+
*/
|
|
160
|
+
async gold(traceId, answer, reason) {
|
|
161
|
+
return this.signal(traceId, { verdict: 'gold', correction: answer, reason });
|
|
162
|
+
}
|
|
119
163
|
/** Every verdict on one conversation, oldest first. */
|
|
120
164
|
async listSignals(traceId) {
|
|
121
165
|
const res = await this._http.fetchGet(`/api/loop/traces/${encodeURIComponent(traceId)}/signals`);
|
package/dist/types.d.ts
CHANGED
|
@@ -1960,7 +1960,15 @@ export type LoopMethod = 'sft' | 'dpo' | 'grpo';
|
|
|
1960
1960
|
/** Who judged an answer. */
|
|
1961
1961
|
export type LoopSource = 'human' | 'verifier' | 'judge' | 'behavioural';
|
|
1962
1962
|
/** The judgement itself. */
|
|
1963
|
-
|
|
1963
|
+
/**
|
|
1964
|
+
* The judgement recorded on an answer.
|
|
1965
|
+
*
|
|
1966
|
+
* `accepted` / `rejected` are the thumbs (see `upvote` / `downvote`).
|
|
1967
|
+
* `edited` says the model was WRONG and carries the better answer.
|
|
1968
|
+
* `gold` records the reference answer for the question, making no claim about
|
|
1969
|
+
* whether the model was right — which is why it is separate from `edited`.
|
|
1970
|
+
*/
|
|
1971
|
+
export type LoopVerdict = 'accepted' | 'rejected' | 'edited' | 'scored' | 'gold';
|
|
1964
1972
|
/** A function an assistant turn asked to invoke. `arguments` is JSON-encoded. */
|
|
1965
1973
|
export interface LoopToolCall {
|
|
1966
1974
|
id?: string;
|