@butlerbot/sdk 0.0.46 → 0.0.48
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/config.d.ts +4 -0
- package/dist/config.js +5 -1
- package/dist/index.d.ts +11 -0
- package/dist/index.js +13 -1
- package/dist/modules/conversation.d.ts +10 -1
- package/dist/modules/judge.d.ts +32 -0
- package/dist/modules/judge.js +29 -0
- package/dist/modules/transport.d.ts +8 -0
- package/dist/modules/transport.js +24 -0
- package/dist/modules/transport_link.js +6 -1
- package/dist/modules/transport_sse.js +3 -0
- package/dist/types/conversation/v3/conversation_v3.d.ts +8 -0
- package/dist/types/judge/index.d.ts +1 -0
- package/dist/types/judge/index.js +17 -0
- package/dist/types/judge/judge.d.ts +103 -0
- package/dist/types/judge/judge.js +12 -0
- package/dist/types/response/v5/ai_response_v5.d.ts +8 -0
- package/dist/types/type_registry.d.ts +1 -0
- package/dist/types/type_registry.js +1 -0
- package/package.json +1 -1
- package/readme.md +55 -2
package/dist/config.d.ts
CHANGED
|
@@ -69,6 +69,10 @@ export declare const CONFIG: {
|
|
|
69
69
|
/** The inbox. One delivery's answer is `${base}/${deliveryId}/answer`. */
|
|
70
70
|
base: string;
|
|
71
71
|
};
|
|
72
|
+
judge: {
|
|
73
|
+
/** A decision on stated facts, by a decision model. */
|
|
74
|
+
base: string;
|
|
75
|
+
};
|
|
72
76
|
};
|
|
73
77
|
};
|
|
74
78
|
export type APIPath = keyof typeof CONFIG.paths.conversation;
|
package/dist/config.js
CHANGED
|
@@ -60,7 +60,11 @@ exports.CONFIG = {
|
|
|
60
60
|
outreach: {
|
|
61
61
|
/** The inbox. One delivery's answer is `${base}/${deliveryId}/answer`. */
|
|
62
62
|
base: "/api/outreach",
|
|
63
|
-
}
|
|
63
|
+
},
|
|
64
|
+
judge: {
|
|
65
|
+
/** A decision on stated facts, by a decision model. */
|
|
66
|
+
base: "/api/judge",
|
|
67
|
+
},
|
|
64
68
|
}
|
|
65
69
|
};
|
|
66
70
|
const withoutTrailingSlash = (url) => url.replace(/\/+$/, "");
|
package/dist/index.d.ts
CHANGED
|
@@ -4,6 +4,8 @@ import { Conversation, ConversationOptions } from "./modules/conversation";
|
|
|
4
4
|
import { UsagePolicyDataOptions } from "./modules/usage";
|
|
5
5
|
import { type CancelJobOptions, type GetJobJournalOptions, type GetJobOptions, type SetPhaseModelOptions, type ListJobsOptions, type ResumeJobOptions, type UpdateJobOptions, type UpdateJobSettingsOptions } from "./modules/jobs";
|
|
6
6
|
import { type AnswerDeliveryOptions, type ListDeliveriesOptions } from "./modules/outreach";
|
|
7
|
+
import { type JudgeOptions } from "./modules/judge";
|
|
8
|
+
import type { JudgeQuestions } from "./types/judge";
|
|
7
9
|
type OptionalApiKey<T> = Omit<T, "apiKey"> & {
|
|
8
10
|
/** Optional API key, defaults to API key specified in client */
|
|
9
11
|
apiKey?: string;
|
|
@@ -64,6 +66,13 @@ export declare class ButlerBotClient {
|
|
|
64
66
|
listDeliveries(config?: OptionalApiKey<ListDeliveriesOptions>): Promise<import("./types/type_registry").DeliveryListResponse>;
|
|
65
67
|
/** Answers a delivery. Throws a `ButlerBotAPIError` with `isConflict` when it was already answered */
|
|
66
68
|
answerDelivery(config: OptionalApiKey<AnswerDeliveryOptions>): Promise<import("./types/type_registry").DeliveryAnswerResponse>;
|
|
69
|
+
/**
|
|
70
|
+
* Asks the judge: a yes/no, pick-one or score decision on facts you state, by a decision
|
|
71
|
+
* model, in well under a second for a fraction of a cent. Answers come back keyed by
|
|
72
|
+
* question id and typed by the question. Throws a `ButlerBotAPIError` with `isBadRequest`
|
|
73
|
+
* on a question the judge cannot ask, carrying its message
|
|
74
|
+
*/
|
|
75
|
+
judge<Q extends JudgeQuestions>(config: OptionalApiKey<JudgeOptions<Q>>): Promise<import("./types/judge").JudgeVerdict<Q>>;
|
|
67
76
|
/** The client's own server and key underneath whatever the call named itself. */
|
|
68
77
|
private forRequest;
|
|
69
78
|
}
|
|
@@ -78,6 +87,8 @@ export { listJobs, getJob, cancelJob, resumeJob, updateJob, updateJobSettings }
|
|
|
78
87
|
export type { JobsRequestOptions, ListJobsOptions, GetJobOptions, CancelJobOptions, ResumeJobOptions, UpdateJobOptions, UpdateJobSettingsOptions, } from "./modules/jobs";
|
|
79
88
|
export { listDeliveries, answerDelivery } from "./modules/outreach";
|
|
80
89
|
export type { OutreachRequestOptions, ListDeliveriesOptions, AnswerDeliveryOptions, } from "./modules/outreach";
|
|
90
|
+
export { judge } from "./modules/judge";
|
|
91
|
+
export type { JudgeRequestOptions, JudgeOptions } from "./modules/judge";
|
|
81
92
|
export { LinkConversationTransport } from "./modules/transport_link";
|
|
82
93
|
export { SSEConversationTransport } from "./modules/transport_sse";
|
|
83
94
|
export type { SteerResult, TurnStopMode, TurnStopped } from "./modules/transport";
|
package/dist/index.js
CHANGED
|
@@ -14,7 +14,7 @@ var __exportStar = (this && this.__exportStar) || function(m, exports) {
|
|
|
14
14
|
for (var p in m) if (p !== "default" && !Object.prototype.hasOwnProperty.call(exports, p)) __createBinding(exports, m, p);
|
|
15
15
|
};
|
|
16
16
|
Object.defineProperty(exports, "__esModule", { value: true });
|
|
17
|
-
exports.RESERVED_TURN_FIELDS = exports.SSEConversationTransport = exports.LinkConversationTransport = exports.answerDelivery = exports.listDeliveries = exports.updateJobSettings = exports.updateJob = exports.resumeJob = exports.cancelJob = exports.getJob = exports.listJobs = exports.ButlerBotAPIError = exports.Conversation = exports.ButlerBotClient = void 0;
|
|
17
|
+
exports.RESERVED_TURN_FIELDS = exports.SSEConversationTransport = exports.LinkConversationTransport = exports.judge = exports.answerDelivery = exports.listDeliveries = exports.updateJobSettings = exports.updateJob = exports.resumeJob = exports.cancelJob = exports.getJob = exports.listJobs = exports.ButlerBotAPIError = exports.Conversation = exports.ButlerBotClient = void 0;
|
|
18
18
|
const config_1 = require("./config");
|
|
19
19
|
const link_1 = require("./link");
|
|
20
20
|
const conversation_1 = require("./modules/conversation");
|
|
@@ -22,6 +22,7 @@ Object.defineProperty(exports, "Conversation", { enumerable: true, get: function
|
|
|
22
22
|
const usage_1 = require("./modules/usage");
|
|
23
23
|
const jobs_1 = require("./modules/jobs");
|
|
24
24
|
const outreach_1 = require("./modules/outreach");
|
|
25
|
+
const judge_1 = require("./modules/judge");
|
|
25
26
|
/**
|
|
26
27
|
* The options a caller actually gave, with the keys they left out removed.
|
|
27
28
|
*
|
|
@@ -112,6 +113,15 @@ class ButlerBotClient {
|
|
|
112
113
|
answerDelivery(config) {
|
|
113
114
|
return (0, outreach_1.answerDelivery)(this.forRequest(config));
|
|
114
115
|
}
|
|
116
|
+
/**
|
|
117
|
+
* Asks the judge: a yes/no, pick-one or score decision on facts you state, by a decision
|
|
118
|
+
* model, in well under a second for a fraction of a cent. Answers come back keyed by
|
|
119
|
+
* question id and typed by the question. Throws a `ButlerBotAPIError` with `isBadRequest`
|
|
120
|
+
* on a question the judge cannot ask, carrying its message
|
|
121
|
+
*/
|
|
122
|
+
judge(config) {
|
|
123
|
+
return (0, judge_1.judge)(this.forRequest(config));
|
|
124
|
+
}
|
|
115
125
|
/** The client's own server and key underneath whatever the call named itself. */
|
|
116
126
|
forRequest(config) {
|
|
117
127
|
return { serverURL: this.serverUrl, apiKey: this.apiKey, debug: this.debug, ...given(config) };
|
|
@@ -133,6 +143,8 @@ Object.defineProperty(exports, "updateJobSettings", { enumerable: true, get: fun
|
|
|
133
143
|
var outreach_2 = require("./modules/outreach");
|
|
134
144
|
Object.defineProperty(exports, "listDeliveries", { enumerable: true, get: function () { return outreach_2.listDeliveries; } });
|
|
135
145
|
Object.defineProperty(exports, "answerDelivery", { enumerable: true, get: function () { return outreach_2.answerDelivery; } });
|
|
146
|
+
var judge_2 = require("./modules/judge");
|
|
147
|
+
Object.defineProperty(exports, "judge", { enumerable: true, get: function () { return judge_2.judge; } });
|
|
136
148
|
var transport_link_1 = require("./modules/transport_link");
|
|
137
149
|
Object.defineProperty(exports, "LinkConversationTransport", { enumerable: true, get: function () { return transport_link_1.LinkConversationTransport; } });
|
|
138
150
|
var transport_sse_1 = require("./modules/transport_sse");
|
|
@@ -36,6 +36,15 @@ export type DialogueRequestParams = {
|
|
|
36
36
|
* 500 characters; the server refuses a longer one. Left off when blank.
|
|
37
37
|
*/
|
|
38
38
|
wake?: string;
|
|
39
|
+
/**
|
|
40
|
+
* Tools switched on for this one turn, by id, that the turn could reach but has off unless
|
|
41
|
+
* asked for: the Discord bot asks for `react` on a turn the summoning judge started, so
|
|
42
|
+
* Alfred may answer a message with a reaction rather than words. Activation, never a grant:
|
|
43
|
+
* the server switches on only what the turn's platform, the user's plan and the user's own
|
|
44
|
+
* settings already allow, and refuses an id it does not know with the nearest real ones.
|
|
45
|
+
* Passed per `send()`/`ask()`, since it is true of the one turn; left off when empty.
|
|
46
|
+
*/
|
|
47
|
+
tools?: string[];
|
|
39
48
|
/**
|
|
40
49
|
* What the platform attaches beside this turn's message, never inside it: what it replies
|
|
41
50
|
* to, who it mentions, where it was sent. See `MessageContextItem` for the shape and the
|
|
@@ -55,7 +64,7 @@ export type DialogueRequestParams = {
|
|
|
55
64
|
*
|
|
56
65
|
* A key the transport already sends (`RESERVED_TURN_FIELDS`: `message`, `chatId`,
|
|
57
66
|
* `api_key`, `model`, `instructions`, `platform`, `address`, `personality`, `wake`,
|
|
58
|
-
* `context`) is refused: `send()` throws, and `ask()` rejects, before anything is sent.
|
|
67
|
+
* `tools`, `context`) is refused: `send()` throws, and `ask()` rejects, before anything is sent.
|
|
59
68
|
*/
|
|
60
69
|
extra?: Record<string, string>;
|
|
61
70
|
};
|
|
@@ -0,0 +1,32 @@
|
|
|
1
|
+
import type { JudgeQuestions, JudgeState, JudgeVerdict } from "../types/judge";
|
|
2
|
+
/** What a judge call needs: where the server is, and who is asking. */
|
|
3
|
+
export type JudgeRequestOptions = {
|
|
4
|
+
serverURL?: string;
|
|
5
|
+
/** The path of the judge route, when it is not the default. */
|
|
6
|
+
path?: string;
|
|
7
|
+
apiKey: string;
|
|
8
|
+
debug?: boolean;
|
|
9
|
+
};
|
|
10
|
+
export type JudgeOptions<Q extends JudgeQuestions = JudgeQuestions> = JudgeRequestOptions & {
|
|
11
|
+
/**
|
|
12
|
+
* The facts the questions are answered from: the message, who sent it, the rules, numbers
|
|
13
|
+
* and dates already worked out in code. Never a transcript or an argument for an answer.
|
|
14
|
+
*/
|
|
15
|
+
state: JudgeState;
|
|
16
|
+
/** The questions, by the id the answer is read back under. All are judged against the same state in one call. */
|
|
17
|
+
questions: Q;
|
|
18
|
+
};
|
|
19
|
+
/**
|
|
20
|
+
* Asks the judge: a yes/no, pick-one or score decision on the facts given, made by a decision
|
|
21
|
+
* model in well under a second for a fraction of a cent.
|
|
22
|
+
*
|
|
23
|
+
* The answers come back keyed by question id and typed by the question: a boolean's `value` and
|
|
24
|
+
* `probability`, a choice's `choice` (one of its criteria's keys), `confidence` and
|
|
25
|
+
* `probabilities`, a score's `level`, `score` and `confidence`. The model writes no text and
|
|
26
|
+
* gives no reason.
|
|
27
|
+
*
|
|
28
|
+
* A question the judge cannot ask (a boolean missing one side, a choice with one label, a score
|
|
29
|
+
* with more than ten levels) is a 400 carrying the judge's own message, and a judgement not
|
|
30
|
+
* reached is a 503: the call throws a `ButlerBotAPIError` either way, never a guess.
|
|
31
|
+
*/
|
|
32
|
+
export declare function judge<Q extends JudgeQuestions>(options: JudgeOptions<Q>): Promise<JudgeVerdict<Q>>;
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
3
|
+
exports.judge = judge;
|
|
4
|
+
const config_1 = require("../config");
|
|
5
|
+
const url_formatter_1 = require("../util/url_formatter");
|
|
6
|
+
const api_request_1 = require("./api_request");
|
|
7
|
+
/**
|
|
8
|
+
* Asks the judge: a yes/no, pick-one or score decision on the facts given, made by a decision
|
|
9
|
+
* model in well under a second for a fraction of a cent.
|
|
10
|
+
*
|
|
11
|
+
* The answers come back keyed by question id and typed by the question: a boolean's `value` and
|
|
12
|
+
* `probability`, a choice's `choice` (one of its criteria's keys), `confidence` and
|
|
13
|
+
* `probabilities`, a score's `level`, `score` and `confidence`. The model writes no text and
|
|
14
|
+
* gives no reason.
|
|
15
|
+
*
|
|
16
|
+
* A question the judge cannot ask (a boolean missing one side, a choice with one label, a score
|
|
17
|
+
* with more than ten levels) is a 400 carrying the judge's own message, and a judgement not
|
|
18
|
+
* reached is a 503: the call throws a `ButlerBotAPIError` either way, never a guess.
|
|
19
|
+
*/
|
|
20
|
+
async function judge(options) {
|
|
21
|
+
const url = (0, url_formatter_1.formatURL)((options.serverURL || config_1.CONFIG.server) + (options.path || config_1.CONFIG.paths.judge.base), {}, { apiKey: options.apiKey, debug: options.debug });
|
|
22
|
+
const response = await (0, api_request_1.requestAPI)({
|
|
23
|
+
url,
|
|
24
|
+
method: "POST",
|
|
25
|
+
body: { state: options.state, questions: options.questions },
|
|
26
|
+
action: "ask the judge",
|
|
27
|
+
});
|
|
28
|
+
return { answers: response.answers, model: response.model, costUsd: response.costUsd };
|
|
29
|
+
}
|
|
@@ -61,6 +61,8 @@ export type TransportTurnRequest = {
|
|
|
61
61
|
personality?: string;
|
|
62
62
|
/** Why Alfred is speaking on this one turn: a line for its system prompt, at most 500 characters. */
|
|
63
63
|
wake?: string;
|
|
64
|
+
/** Tools switched on for this one turn, by id. Sent only when non-empty; see `DialogueRequestParams.tools`. */
|
|
65
|
+
tools?: string[];
|
|
64
66
|
/** What the platform attaches beside this turn's message. Sent only when non-empty. */
|
|
65
67
|
context?: MessageContextItem[];
|
|
66
68
|
/** Further turn parameters, forwarded to the server verbatim. See `DialogueRequestParams.extra`. */
|
|
@@ -81,6 +83,12 @@ export declare const RESERVED_TURN_FIELDS: readonly string[];
|
|
|
81
83
|
export declare function turnExtra(extra: Record<string, string> | undefined): Record<string, string> | undefined;
|
|
82
84
|
/** A turn's context as it travels, JSON-encoded, or nothing when there is none. */
|
|
83
85
|
export declare function turnContext(context: MessageContextItem[] | undefined): string | undefined;
|
|
86
|
+
/**
|
|
87
|
+
* A turn's tools as they travel: the ids trimmed, emptied ones dropped, repeats folded, joined
|
|
88
|
+
* with commas; nothing when none is left. A comma in an id would read as two ids, so one is
|
|
89
|
+
* refused before anything is sent.
|
|
90
|
+
*/
|
|
91
|
+
export declare function turnTools(tools: string[] | undefined): string | undefined;
|
|
84
92
|
export type TransportHandlers = {
|
|
85
93
|
/** One payload of the stream, already in the shape callers expect. */
|
|
86
94
|
payload(payload: unknown): void;
|
|
@@ -11,6 +11,7 @@ Object.defineProperty(exports, "__esModule", { value: true });
|
|
|
11
11
|
exports.RESERVED_TURN_FIELDS = void 0;
|
|
12
12
|
exports.turnExtra = turnExtra;
|
|
13
13
|
exports.turnContext = turnContext;
|
|
14
|
+
exports.turnTools = turnTools;
|
|
14
15
|
exports.noticePayload = noticePayload;
|
|
15
16
|
exports.convoStartedPayload = convoStartedPayload;
|
|
16
17
|
exports.completedPayload = completedPayload;
|
|
@@ -31,6 +32,7 @@ exports.RESERVED_TURN_FIELDS = [
|
|
|
31
32
|
"address",
|
|
32
33
|
"personality",
|
|
33
34
|
"wake",
|
|
35
|
+
"tools",
|
|
34
36
|
"context",
|
|
35
37
|
];
|
|
36
38
|
/**
|
|
@@ -60,6 +62,28 @@ function turnExtra(extra) {
|
|
|
60
62
|
function turnContext(context) {
|
|
61
63
|
return context && context.length > 0 ? JSON.stringify(context) : undefined;
|
|
62
64
|
}
|
|
65
|
+
/**
|
|
66
|
+
* A turn's tools as they travel: the ids trimmed, emptied ones dropped, repeats folded, joined
|
|
67
|
+
* with commas; nothing when none is left. A comma in an id would read as two ids, so one is
|
|
68
|
+
* refused before anything is sent.
|
|
69
|
+
*/
|
|
70
|
+
function turnTools(tools) {
|
|
71
|
+
if (!tools || tools.length === 0)
|
|
72
|
+
return undefined;
|
|
73
|
+
const ids = [];
|
|
74
|
+
for (const raw of tools) {
|
|
75
|
+
if (typeof raw !== "string")
|
|
76
|
+
throw new Error(`A turn's \`tools\` are tool ids; one is ${typeof raw}.`);
|
|
77
|
+
const id = raw.trim();
|
|
78
|
+
if (!id)
|
|
79
|
+
continue;
|
|
80
|
+
if (id.includes(","))
|
|
81
|
+
throw new Error(`A turn's \`tools\` entry may not contain a comma: "${id}".`);
|
|
82
|
+
if (!ids.includes(id))
|
|
83
|
+
ids.push(id);
|
|
84
|
+
}
|
|
85
|
+
return ids.length > 0 ? ids.join(",") : undefined;
|
|
86
|
+
}
|
|
63
87
|
// =============================================
|
|
64
88
|
// PAYLOAD SHAPES
|
|
65
89
|
// =============================================
|
|
@@ -179,7 +179,12 @@ class LinkConversationTransport {
|
|
|
179
179
|
};
|
|
180
180
|
learnChatId(progress.chatId);
|
|
181
181
|
const context = (0, transport_1.turnContext)(request.context);
|
|
182
|
-
|
|
182
|
+
// The turn's tools ride in `extra`, which the Link service spreads into the same query
|
|
183
|
+
// the SSE transport builds: the server reads `tools` either way, and a caller's own
|
|
184
|
+
// `extra` can never carry that key, since the transport reserves it.
|
|
185
|
+
const tools = (0, transport_1.turnTools)(request.tools);
|
|
186
|
+
const own = (0, transport_1.turnExtra)(request.extra);
|
|
187
|
+
const extra = tools ? { ...(own ?? {}), tools } : own;
|
|
183
188
|
let done;
|
|
184
189
|
try {
|
|
185
190
|
done = await this.link.exchange("conversation.chat", {
|
|
@@ -124,6 +124,9 @@ function asQuery(request) {
|
|
|
124
124
|
const wake = request.wake?.trim();
|
|
125
125
|
if (wake)
|
|
126
126
|
query.wake = wake;
|
|
127
|
+
const tools = (0, transport_1.turnTools)(request.tools);
|
|
128
|
+
if (tools)
|
|
129
|
+
query.tools = tools;
|
|
127
130
|
const context = (0, transport_1.turnContext)(request.context);
|
|
128
131
|
if (context)
|
|
129
132
|
query.context = context;
|
|
@@ -52,6 +52,14 @@ export type ToolStatus = {
|
|
|
52
52
|
content?: ToolStatusContent[];
|
|
53
53
|
/** Whether this tool call is ending the conversation */
|
|
54
54
|
endingConvo?: boolean;
|
|
55
|
+
/** The tool that ran: its id, or a raw (MCP) tool's key. A fact for clients to present, never a label. */
|
|
56
|
+
toolId?: string;
|
|
57
|
+
/** Epoch ms of the first status emitted for this id; the same on every later status for it. */
|
|
58
|
+
startedAt?: number;
|
|
59
|
+
/** Epoch ms at which this id reached `completed` or `failed`. */
|
|
60
|
+
endedAt?: number;
|
|
61
|
+
/** The id of an earlier failed call of the same tool, in the same turn, that this call retries. */
|
|
62
|
+
retryOf?: string;
|
|
55
63
|
};
|
|
56
64
|
export type TokenUsage = {
|
|
57
65
|
/**
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export * from "./judge";
|
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
|
|
3
|
+
if (k2 === undefined) k2 = k;
|
|
4
|
+
var desc = Object.getOwnPropertyDescriptor(m, k);
|
|
5
|
+
if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
|
|
6
|
+
desc = { enumerable: true, get: function() { return m[k]; } };
|
|
7
|
+
}
|
|
8
|
+
Object.defineProperty(o, k2, desc);
|
|
9
|
+
}) : (function(o, m, k, k2) {
|
|
10
|
+
if (k2 === undefined) k2 = k;
|
|
11
|
+
o[k2] = m[k];
|
|
12
|
+
}));
|
|
13
|
+
var __exportStar = (this && this.__exportStar) || function(m, exports) {
|
|
14
|
+
for (var p in m) if (p !== "default" && !Object.prototype.hasOwnProperty.call(exports, p)) __createBinding(exports, m, p);
|
|
15
|
+
};
|
|
16
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
17
|
+
__exportStar(require("./judge"), exports);
|
|
@@ -0,0 +1,103 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* The judge: a yes/no, pick-one or score decision made by a decision model, on facts the caller
|
|
3
|
+
* states.
|
|
4
|
+
*
|
|
5
|
+
* Not a conversation. It writes no text and gives no reason: one call, every question judged
|
|
6
|
+
* against the same state, and typed answers back with probabilities. The state is facts (the
|
|
7
|
+
* message, who sent it, the rules, numbers and dates already worked out in code), never a
|
|
8
|
+
* transcript or an argument for an answer. State and the longest question together fit in about
|
|
9
|
+
* 32,000 tokens.
|
|
10
|
+
*/
|
|
11
|
+
/**
|
|
12
|
+
* A yes/no question.
|
|
13
|
+
*
|
|
14
|
+
* Both sides are described, because the model weighs the state against each description rather
|
|
15
|
+
* than against the question alone, and the server refuses a question that leaves one out.
|
|
16
|
+
*/
|
|
17
|
+
export type JudgeBooleanQuestion = {
|
|
18
|
+
type: "boolean";
|
|
19
|
+
/** What is being decided about the state, in one or two sentences. */
|
|
20
|
+
instructions: string;
|
|
21
|
+
criteria: {
|
|
22
|
+
/** What a yes looks like in the state. */
|
|
23
|
+
true: string;
|
|
24
|
+
/** What a no looks like in the state. */
|
|
25
|
+
false: string;
|
|
26
|
+
};
|
|
27
|
+
/** The probability of yes (between 0 and 1, exclusive) at which `value` is true. 0.5 when unset. */
|
|
28
|
+
threshold?: number;
|
|
29
|
+
};
|
|
30
|
+
/**
|
|
31
|
+
* Pick one label.
|
|
32
|
+
*
|
|
33
|
+
* Every label the answer may be is a key of `criteria`, described by its value; at least two.
|
|
34
|
+
* There is no automatic "none": when the state may fit no label, add one ("none": "Nothing above
|
|
35
|
+
* fits"), or the nearest label is picked however poor the fit.
|
|
36
|
+
*/
|
|
37
|
+
export type JudgeChoiceQuestion<L extends string = string> = {
|
|
38
|
+
type: "choice";
|
|
39
|
+
instructions: string;
|
|
40
|
+
criteria: Record<L, string>;
|
|
41
|
+
};
|
|
42
|
+
/**
|
|
43
|
+
* A place on an ordered scale.
|
|
44
|
+
*
|
|
45
|
+
* `criteria[0]` describes the bottom of the scale and each entry after it one level up; at most
|
|
46
|
+
* ten levels.
|
|
47
|
+
*/
|
|
48
|
+
export type JudgeScoreQuestion = {
|
|
49
|
+
type: "score";
|
|
50
|
+
instructions: string;
|
|
51
|
+
criteria: string[];
|
|
52
|
+
};
|
|
53
|
+
export type JudgeQuestion = JudgeBooleanQuestion | JudgeChoiceQuestion | JudgeScoreQuestion;
|
|
54
|
+
/** The questions of one call, by an id (1-64 letters, digits, `_` or `-`) the answer is read back under. */
|
|
55
|
+
export type JudgeQuestions = Record<string, JudgeQuestion>;
|
|
56
|
+
/** The facts the questions are answered from: text, or an object the server renders. */
|
|
57
|
+
export type JudgeState = string | Record<string, unknown>;
|
|
58
|
+
export type JudgeBooleanAnswer = {
|
|
59
|
+
type: "boolean";
|
|
60
|
+
/** True when `probability` reached the question's threshold. */
|
|
61
|
+
value: boolean;
|
|
62
|
+
/** How likely yes is, 0 to 1. */
|
|
63
|
+
probability: number;
|
|
64
|
+
};
|
|
65
|
+
export type JudgeChoiceAnswer<L extends string = string> = {
|
|
66
|
+
type: "choice";
|
|
67
|
+
/** The label picked: one of the question's criteria keys. */
|
|
68
|
+
choice: L;
|
|
69
|
+
/** How sure, 0 to 1. */
|
|
70
|
+
confidence: number;
|
|
71
|
+
/** Every label's probability, keyed by label. */
|
|
72
|
+
probabilities?: Partial<Record<L, number>>;
|
|
73
|
+
};
|
|
74
|
+
export type JudgeScoreAnswer = {
|
|
75
|
+
type: "score";
|
|
76
|
+
/** The probability-weighted mean of the levels, 0-based; a fraction when the model is torn. */
|
|
77
|
+
score: number;
|
|
78
|
+
/** The single most likely level, 0-based: an index into the question's `criteria`. */
|
|
79
|
+
level: number;
|
|
80
|
+
/** How sure of `level`, 0 to 1. */
|
|
81
|
+
confidence: number;
|
|
82
|
+
/** Each level's probability, keyed by its index as a string. */
|
|
83
|
+
probabilities?: Record<string, number>;
|
|
84
|
+
};
|
|
85
|
+
export type JudgeAnswer = JudgeBooleanAnswer | JudgeChoiceAnswer | JudgeScoreAnswer;
|
|
86
|
+
/** The answer a question gets, typed by the question: a choice's label is one of its criteria's keys. */
|
|
87
|
+
export type JudgeAnswerFor<Q extends JudgeQuestion> = Q extends JudgeBooleanQuestion ? JudgeBooleanAnswer : Q extends JudgeChoiceQuestion ? JudgeChoiceAnswer<Extract<keyof Q["criteria"], string>> : Q extends JudgeScoreQuestion ? JudgeScoreAnswer : never;
|
|
88
|
+
export type JudgeAnswers<Q extends JudgeQuestions> = {
|
|
89
|
+
[K in keyof Q]: JudgeAnswerFor<Q[K]>;
|
|
90
|
+
};
|
|
91
|
+
/** What a decision comes back as. */
|
|
92
|
+
export type JudgeVerdict<Q extends JudgeQuestions = JudgeQuestions> = {
|
|
93
|
+
/** One answer per question, under the question's id. */
|
|
94
|
+
answers: JudgeAnswers<Q>;
|
|
95
|
+
/** The decision model that answered. */
|
|
96
|
+
model: string;
|
|
97
|
+
/** What the judgement cost, USD. It is already on the account's ledger. */
|
|
98
|
+
costUsd: number;
|
|
99
|
+
};
|
|
100
|
+
/** The route's whole answer. */
|
|
101
|
+
export type JudgeResponse<Q extends JudgeQuestions = JudgeQuestions> = JudgeVerdict<Q> & {
|
|
102
|
+
success: true;
|
|
103
|
+
};
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
"use strict";
|
|
2
|
+
/**
|
|
3
|
+
* The judge: a yes/no, pick-one or score decision made by a decision model, on facts the caller
|
|
4
|
+
* states.
|
|
5
|
+
*
|
|
6
|
+
* Not a conversation. It writes no text and gives no reason: one call, every question judged
|
|
7
|
+
* against the same state, and typed answers back with probabilities. The state is facts (the
|
|
8
|
+
* message, who sent it, the rules, numbers and dates already worked out in code), never a
|
|
9
|
+
* transcript or an argument for an answer. State and the longest question together fit in about
|
|
10
|
+
* 32,000 tokens.
|
|
11
|
+
*/
|
|
12
|
+
Object.defineProperty(exports, "__esModule", { value: true });
|
|
@@ -39,6 +39,14 @@ export type ToolStatus = {
|
|
|
39
39
|
content?: ToolStatusContent[];
|
|
40
40
|
/** Whether this tool call is ending the conversation */
|
|
41
41
|
endingConvo?: boolean;
|
|
42
|
+
/** The tool that ran: its id, or a raw (MCP) tool's key. A fact for clients to present, never a label. */
|
|
43
|
+
toolId?: string;
|
|
44
|
+
/** Epoch ms of the first status emitted for this id; the same on every later status for it. */
|
|
45
|
+
startedAt?: number;
|
|
46
|
+
/** Epoch ms at which this id reached `completed` or `failed`. */
|
|
47
|
+
endedAt?: number;
|
|
48
|
+
/** The id of an earlier failed call of the same tool, in the same turn, that this call retries. */
|
|
49
|
+
retryOf?: string;
|
|
42
50
|
};
|
|
43
51
|
export type BaseResponseMetadata = {
|
|
44
52
|
/** Unique response ID */
|
|
@@ -23,4 +23,5 @@ __exportStar(require("./conversation/address"), exports);
|
|
|
23
23
|
__exportStar(require("./conversation/context"), exports);
|
|
24
24
|
__exportStar(require("./jobs"), exports);
|
|
25
25
|
__exportStar(require("./outreach"), exports);
|
|
26
|
+
__exportStar(require("./judge"), exports);
|
|
26
27
|
__exportStar(require("./error"), exports);
|
package/package.json
CHANGED
package/readme.md
CHANGED
|
@@ -387,10 +387,63 @@ one, and answering it is refused with a 409 just as one already answered is. `so
|
|
|
387
387
|
wrote a delivery: `shift` for a question a job's shift asked, `runtime` for the job's own
|
|
388
388
|
status tells, `gate` for an approval the autonomy gate asked for.
|
|
389
389
|
|
|
390
|
+
## Judge
|
|
391
|
+
|
|
392
|
+
A yes/no, pick-one or score decision on facts you state, made by a decision model: well under
|
|
393
|
+
a second and a fraction of a cent, so an app can ask it per event. It is Alfred's own judge,
|
|
394
|
+
the one his jobs and reflexes decide with, for your code to decide with too.
|
|
395
|
+
|
|
396
|
+
```typescript
|
|
397
|
+
const { answers, model, costUsd } = await client.judge({
|
|
398
|
+
state: { message: text, author: "bot: false", channel: "general", rules },
|
|
399
|
+
questions: {
|
|
400
|
+
hostile: {
|
|
401
|
+
type: "boolean",
|
|
402
|
+
instructions: "Is the message hostile?",
|
|
403
|
+
criteria: { true: "It attacks or insults someone", false: "It is civil, however blunt" },
|
|
404
|
+
},
|
|
405
|
+
rule: {
|
|
406
|
+
type: "choice",
|
|
407
|
+
instructions: "Which server rule does the message break?",
|
|
408
|
+
criteria: { spam: "Promotional or repeated", abuse: "Insults a person", none: "No rule is broken" },
|
|
409
|
+
},
|
|
410
|
+
heat: {
|
|
411
|
+
type: "score",
|
|
412
|
+
instructions: "How heated is the message?",
|
|
413
|
+
criteria: ["Calm", "Annoyed", "Furious"],
|
|
414
|
+
},
|
|
415
|
+
},
|
|
416
|
+
});
|
|
417
|
+
|
|
418
|
+
answers.hostile.value; // boolean: value, probability
|
|
419
|
+
answers.rule.choice; // "spam" | "abuse" | "none", with confidence and every label's probability
|
|
420
|
+
answers.heat.level; // 0..2, with score (the weighted mean) and confidence
|
|
421
|
+
```
|
|
422
|
+
|
|
423
|
+
Every question is judged against the same `state` in one call, and the answers come back keyed
|
|
424
|
+
by question id and typed by the question: a choice's `choice` is one of its own criteria keys.
|
|
425
|
+
It is a decision model, not a chat model: it writes no text and does no reasoning, so
|
|
426
|
+
|
|
427
|
+
- state facts, never a transcript or an argument for an answer;
|
|
428
|
+
- work out numbers and dates in code and state the result;
|
|
429
|
+
- describe both sides of a boolean, since the model weighs the state against each;
|
|
430
|
+
- add a `none` label to a choice when nothing may fit, or the nearest label is picked however
|
|
431
|
+
poor the fit;
|
|
432
|
+
- keep a score to ten levels, `criteria[0]` the bottom.
|
|
433
|
+
|
|
434
|
+
State and the longest question together fit in about 32,000 tokens. A question the judge cannot
|
|
435
|
+
ask is a 400 carrying its message, and a judgement it could not reach is a 503: both throw
|
|
436
|
+
`ButlerBotAPIError`, never a guess. Needs `judge.run`.
|
|
437
|
+
|
|
438
|
+
The judge is for a decision your app makes itself. For Alfred to react to something, do not
|
|
439
|
+
judge first: report it as a [hook event](#hooks-emit-or-report-what-matched) and the user's
|
|
440
|
+
reflex decides, with the judge inside it when it needs one. That keeps what Alfred reacts to,
|
|
441
|
+
and what it costs, in the user's hands.
|
|
442
|
+
|
|
390
443
|
### When a call fails
|
|
391
444
|
|
|
392
|
-
Every jobs and
|
|
393
|
-
success. The status is on the error, so the cases worth branching on are told apart without
|
|
445
|
+
Every jobs, outreach and judge call throws `ButlerBotAPIError` when the server does not answer
|
|
446
|
+
with a success. The status is on the error, so the cases worth branching on are told apart without
|
|
394
447
|
reading a message, and the parsed body is kept — a rejected cancel still carries the job, a
|
|
395
448
|
rejected answer still carries the delivery:
|
|
396
449
|
|