@butlerbot/sdk 0.0.46 → 0.0.47

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/dist/config.d.ts CHANGED
@@ -69,6 +69,10 @@ export declare const CONFIG: {
69
69
  /** The inbox. One delivery's answer is `${base}/${deliveryId}/answer`. */
70
70
  base: string;
71
71
  };
72
+ judge: {
73
+ /** A decision on stated facts, by a decision model. */
74
+ base: string;
75
+ };
72
76
  };
73
77
  };
74
78
  export type APIPath = keyof typeof CONFIG.paths.conversation;
package/dist/config.js CHANGED
@@ -60,7 +60,11 @@ exports.CONFIG = {
60
60
  outreach: {
61
61
  /** The inbox. One delivery's answer is `${base}/${deliveryId}/answer`. */
62
62
  base: "/api/outreach",
63
- }
63
+ },
64
+ judge: {
65
+ /** A decision on stated facts, by a decision model. */
66
+ base: "/api/judge",
67
+ },
64
68
  }
65
69
  };
66
70
  const withoutTrailingSlash = (url) => url.replace(/\/+$/, "");
package/dist/index.d.ts CHANGED
@@ -4,6 +4,8 @@ import { Conversation, ConversationOptions } from "./modules/conversation";
4
4
  import { UsagePolicyDataOptions } from "./modules/usage";
5
5
  import { type CancelJobOptions, type GetJobJournalOptions, type GetJobOptions, type SetPhaseModelOptions, type ListJobsOptions, type ResumeJobOptions, type UpdateJobOptions, type UpdateJobSettingsOptions } from "./modules/jobs";
6
6
  import { type AnswerDeliveryOptions, type ListDeliveriesOptions } from "./modules/outreach";
7
+ import { type JudgeOptions } from "./modules/judge";
8
+ import type { JudgeQuestions } from "./types/judge";
7
9
  type OptionalApiKey<T> = Omit<T, "apiKey"> & {
8
10
  /** Optional API key, defaults to API key specified in client */
9
11
  apiKey?: string;
@@ -64,6 +66,13 @@ export declare class ButlerBotClient {
64
66
  listDeliveries(config?: OptionalApiKey<ListDeliveriesOptions>): Promise<import("./types/type_registry").DeliveryListResponse>;
65
67
  /** Answers a delivery. Throws a `ButlerBotAPIError` with `isConflict` when it was already answered */
66
68
  answerDelivery(config: OptionalApiKey<AnswerDeliveryOptions>): Promise<import("./types/type_registry").DeliveryAnswerResponse>;
69
+ /**
70
+ * Asks the judge: a yes/no, pick-one or score decision on facts you state, by a decision
71
+ * model, in well under a second for a fraction of a cent. Answers come back keyed by
72
+ * question id and typed by the question. Throws a `ButlerBotAPIError` with `isBadRequest`
73
+ * on a question the judge cannot ask, carrying its message
74
+ */
75
+ judge<Q extends JudgeQuestions>(config: OptionalApiKey<JudgeOptions<Q>>): Promise<import("./types/judge").JudgeVerdict<Q>>;
67
76
  /** The client's own server and key underneath whatever the call named itself. */
68
77
  private forRequest;
69
78
  }
@@ -78,6 +87,8 @@ export { listJobs, getJob, cancelJob, resumeJob, updateJob, updateJobSettings }
78
87
  export type { JobsRequestOptions, ListJobsOptions, GetJobOptions, CancelJobOptions, ResumeJobOptions, UpdateJobOptions, UpdateJobSettingsOptions, } from "./modules/jobs";
79
88
  export { listDeliveries, answerDelivery } from "./modules/outreach";
80
89
  export type { OutreachRequestOptions, ListDeliveriesOptions, AnswerDeliveryOptions, } from "./modules/outreach";
90
+ export { judge } from "./modules/judge";
91
+ export type { JudgeRequestOptions, JudgeOptions } from "./modules/judge";
81
92
  export { LinkConversationTransport } from "./modules/transport_link";
82
93
  export { SSEConversationTransport } from "./modules/transport_sse";
83
94
  export type { SteerResult, TurnStopMode, TurnStopped } from "./modules/transport";
package/dist/index.js CHANGED
@@ -14,7 +14,7 @@ var __exportStar = (this && this.__exportStar) || function(m, exports) {
14
14
  for (var p in m) if (p !== "default" && !Object.prototype.hasOwnProperty.call(exports, p)) __createBinding(exports, m, p);
15
15
  };
16
16
  Object.defineProperty(exports, "__esModule", { value: true });
17
- exports.RESERVED_TURN_FIELDS = exports.SSEConversationTransport = exports.LinkConversationTransport = exports.answerDelivery = exports.listDeliveries = exports.updateJobSettings = exports.updateJob = exports.resumeJob = exports.cancelJob = exports.getJob = exports.listJobs = exports.ButlerBotAPIError = exports.Conversation = exports.ButlerBotClient = void 0;
17
+ exports.RESERVED_TURN_FIELDS = exports.SSEConversationTransport = exports.LinkConversationTransport = exports.judge = exports.answerDelivery = exports.listDeliveries = exports.updateJobSettings = exports.updateJob = exports.resumeJob = exports.cancelJob = exports.getJob = exports.listJobs = exports.ButlerBotAPIError = exports.Conversation = exports.ButlerBotClient = void 0;
18
18
  const config_1 = require("./config");
19
19
  const link_1 = require("./link");
20
20
  const conversation_1 = require("./modules/conversation");
@@ -22,6 +22,7 @@ Object.defineProperty(exports, "Conversation", { enumerable: true, get: function
22
22
  const usage_1 = require("./modules/usage");
23
23
  const jobs_1 = require("./modules/jobs");
24
24
  const outreach_1 = require("./modules/outreach");
25
+ const judge_1 = require("./modules/judge");
25
26
  /**
26
27
  * The options a caller actually gave, with the keys they left out removed.
27
28
  *
@@ -112,6 +113,15 @@ class ButlerBotClient {
112
113
  answerDelivery(config) {
113
114
  return (0, outreach_1.answerDelivery)(this.forRequest(config));
114
115
  }
116
+ /**
117
+ * Asks the judge: a yes/no, pick-one or score decision on facts you state, by a decision
118
+ * model, in well under a second for a fraction of a cent. Answers come back keyed by
119
+ * question id and typed by the question. Throws a `ButlerBotAPIError` with `isBadRequest`
120
+ * on a question the judge cannot ask, carrying its message
121
+ */
122
+ judge(config) {
123
+ return (0, judge_1.judge)(this.forRequest(config));
124
+ }
115
125
  /** The client's own server and key underneath whatever the call named itself. */
116
126
  forRequest(config) {
117
127
  return { serverURL: this.serverUrl, apiKey: this.apiKey, debug: this.debug, ...given(config) };
@@ -133,6 +143,8 @@ Object.defineProperty(exports, "updateJobSettings", { enumerable: true, get: fun
133
143
  var outreach_2 = require("./modules/outreach");
134
144
  Object.defineProperty(exports, "listDeliveries", { enumerable: true, get: function () { return outreach_2.listDeliveries; } });
135
145
  Object.defineProperty(exports, "answerDelivery", { enumerable: true, get: function () { return outreach_2.answerDelivery; } });
146
+ var judge_2 = require("./modules/judge");
147
+ Object.defineProperty(exports, "judge", { enumerable: true, get: function () { return judge_2.judge; } });
136
148
  var transport_link_1 = require("./modules/transport_link");
137
149
  Object.defineProperty(exports, "LinkConversationTransport", { enumerable: true, get: function () { return transport_link_1.LinkConversationTransport; } });
138
150
  var transport_sse_1 = require("./modules/transport_sse");
@@ -0,0 +1,32 @@
1
+ import type { JudgeQuestions, JudgeState, JudgeVerdict } from "../types/judge";
2
+ /** What a judge call needs: where the server is, and who is asking. */
3
+ export type JudgeRequestOptions = {
4
+ serverURL?: string;
5
+ /** The path of the judge route, when it is not the default. */
6
+ path?: string;
7
+ apiKey: string;
8
+ debug?: boolean;
9
+ };
10
+ export type JudgeOptions<Q extends JudgeQuestions = JudgeQuestions> = JudgeRequestOptions & {
11
+ /**
12
+ * The facts the questions are answered from: the message, who sent it, the rules, numbers
13
+ * and dates already worked out in code. Never a transcript or an argument for an answer.
14
+ */
15
+ state: JudgeState;
16
+ /** The questions, by the id the answer is read back under. All are judged against the same state in one call. */
17
+ questions: Q;
18
+ };
19
+ /**
20
+ * Asks the judge: a yes/no, pick-one or score decision on the facts given, made by a decision
21
+ * model in well under a second for a fraction of a cent.
22
+ *
23
+ * The answers come back keyed by question id and typed by the question: a boolean's `value` and
24
+ * `probability`, a choice's `choice` (one of its criteria's keys), `confidence` and
25
+ * `probabilities`, a score's `level`, `score` and `confidence`. The model writes no text and
26
+ * gives no reason.
27
+ *
28
+ * A question the judge cannot ask (a boolean missing one side, a choice with one label, a score
29
+ * with more than ten levels) is a 400 carrying the judge's own message, and a judgement not
30
+ * reached is a 503: the call throws a `ButlerBotAPIError` either way, never a guess.
31
+ */
32
+ export declare function judge<Q extends JudgeQuestions>(options: JudgeOptions<Q>): Promise<JudgeVerdict<Q>>;
@@ -0,0 +1,29 @@
1
+ "use strict";
2
+ Object.defineProperty(exports, "__esModule", { value: true });
3
+ exports.judge = judge;
4
+ const config_1 = require("../config");
5
+ const url_formatter_1 = require("../util/url_formatter");
6
+ const api_request_1 = require("./api_request");
7
+ /**
8
+ * Asks the judge: a yes/no, pick-one or score decision on the facts given, made by a decision
9
+ * model in well under a second for a fraction of a cent.
10
+ *
11
+ * The answers come back keyed by question id and typed by the question: a boolean's `value` and
12
+ * `probability`, a choice's `choice` (one of its criteria's keys), `confidence` and
13
+ * `probabilities`, a score's `level`, `score` and `confidence`. The model writes no text and
14
+ * gives no reason.
15
+ *
16
+ * A question the judge cannot ask (a boolean missing one side, a choice with one label, a score
17
+ * with more than ten levels) is a 400 carrying the judge's own message, and a judgement not
18
+ * reached is a 503: the call throws a `ButlerBotAPIError` either way, never a guess.
19
+ */
20
+ async function judge(options) {
21
+ const url = (0, url_formatter_1.formatURL)((options.serverURL || config_1.CONFIG.server) + (options.path || config_1.CONFIG.paths.judge.base), {}, { apiKey: options.apiKey, debug: options.debug });
22
+ const response = await (0, api_request_1.requestAPI)({
23
+ url,
24
+ method: "POST",
25
+ body: { state: options.state, questions: options.questions },
26
+ action: "ask the judge",
27
+ });
28
+ return { answers: response.answers, model: response.model, costUsd: response.costUsd };
29
+ }
@@ -52,6 +52,14 @@ export type ToolStatus = {
52
52
  content?: ToolStatusContent[];
53
53
  /** Whether this tool call is ending the conversation */
54
54
  endingConvo?: boolean;
55
+ /** The tool that ran: its id, or a raw (MCP) tool's key. A fact for clients to present, never a label. */
56
+ toolId?: string;
57
+ /** Epoch ms of the first status emitted for this id; the same on every later status for it. */
58
+ startedAt?: number;
59
+ /** Epoch ms at which this id reached `completed` or `failed`. */
60
+ endedAt?: number;
61
+ /** The id of an earlier failed call of the same tool, in the same turn, that this call retries. */
62
+ retryOf?: string;
55
63
  };
56
64
  export type TokenUsage = {
57
65
  /**
@@ -0,0 +1 @@
1
+ export * from "./judge";
@@ -0,0 +1,17 @@
1
+ "use strict";
2
+ var __createBinding = (this && this.__createBinding) || (Object.create ? (function(o, m, k, k2) {
3
+ if (k2 === undefined) k2 = k;
4
+ var desc = Object.getOwnPropertyDescriptor(m, k);
5
+ if (!desc || ("get" in desc ? !m.__esModule : desc.writable || desc.configurable)) {
6
+ desc = { enumerable: true, get: function() { return m[k]; } };
7
+ }
8
+ Object.defineProperty(o, k2, desc);
9
+ }) : (function(o, m, k, k2) {
10
+ if (k2 === undefined) k2 = k;
11
+ o[k2] = m[k];
12
+ }));
13
+ var __exportStar = (this && this.__exportStar) || function(m, exports) {
14
+ for (var p in m) if (p !== "default" && !Object.prototype.hasOwnProperty.call(exports, p)) __createBinding(exports, m, p);
15
+ };
16
+ Object.defineProperty(exports, "__esModule", { value: true });
17
+ __exportStar(require("./judge"), exports);
@@ -0,0 +1,103 @@
1
+ /**
2
+ * The judge: a yes/no, pick-one or score decision made by a decision model, on facts the caller
3
+ * states.
4
+ *
5
+ * Not a conversation. It writes no text and gives no reason: one call, every question judged
6
+ * against the same state, and typed answers back with probabilities. The state is facts (the
7
+ * message, who sent it, the rules, numbers and dates already worked out in code), never a
8
+ * transcript or an argument for an answer. State and the longest question together fit in about
9
+ * 32,000 tokens.
10
+ */
11
+ /**
12
+ * A yes/no question.
13
+ *
14
+ * Both sides are described, because the model weighs the state against each description rather
15
+ * than against the question alone, and the server refuses a question that leaves one out.
16
+ */
17
+ export type JudgeBooleanQuestion = {
18
+ type: "boolean";
19
+ /** What is being decided about the state, in one or two sentences. */
20
+ instructions: string;
21
+ criteria: {
22
+ /** What a yes looks like in the state. */
23
+ true: string;
24
+ /** What a no looks like in the state. */
25
+ false: string;
26
+ };
27
+ /** The probability of yes (between 0 and 1, exclusive) at which `value` is true. 0.5 when unset. */
28
+ threshold?: number;
29
+ };
30
+ /**
31
+ * Pick one label.
32
+ *
33
+ * Every label the answer may be is a key of `criteria`, described by its value; at least two.
34
+ * There is no automatic "none": when the state may fit no label, add one ("none": "Nothing above
35
+ * fits"), or the nearest label is picked however poor the fit.
36
+ */
37
+ export type JudgeChoiceQuestion<L extends string = string> = {
38
+ type: "choice";
39
+ instructions: string;
40
+ criteria: Record<L, string>;
41
+ };
42
+ /**
43
+ * A place on an ordered scale.
44
+ *
45
+ * `criteria[0]` describes the bottom of the scale and each entry after it one level up; at most
46
+ * ten levels.
47
+ */
48
+ export type JudgeScoreQuestion = {
49
+ type: "score";
50
+ instructions: string;
51
+ criteria: string[];
52
+ };
53
+ export type JudgeQuestion = JudgeBooleanQuestion | JudgeChoiceQuestion | JudgeScoreQuestion;
54
+ /** The questions of one call, by an id (1-64 letters, digits, `_` or `-`) the answer is read back under. */
55
+ export type JudgeQuestions = Record<string, JudgeQuestion>;
56
+ /** The facts the questions are answered from: text, or an object the server renders. */
57
+ export type JudgeState = string | Record<string, unknown>;
58
+ export type JudgeBooleanAnswer = {
59
+ type: "boolean";
60
+ /** True when `probability` reached the question's threshold. */
61
+ value: boolean;
62
+ /** How likely yes is, 0 to 1. */
63
+ probability: number;
64
+ };
65
+ export type JudgeChoiceAnswer<L extends string = string> = {
66
+ type: "choice";
67
+ /** The label picked: one of the question's criteria keys. */
68
+ choice: L;
69
+ /** How sure, 0 to 1. */
70
+ confidence: number;
71
+ /** Every label's probability, keyed by label. */
72
+ probabilities?: Partial<Record<L, number>>;
73
+ };
74
+ export type JudgeScoreAnswer = {
75
+ type: "score";
76
+ /** The probability-weighted mean of the levels, 0-based; a fraction when the model is torn. */
77
+ score: number;
78
+ /** The single most likely level, 0-based: an index into the question's `criteria`. */
79
+ level: number;
80
+ /** How sure of `level`, 0 to 1. */
81
+ confidence: number;
82
+ /** Each level's probability, keyed by its index as a string. */
83
+ probabilities?: Record<string, number>;
84
+ };
85
+ export type JudgeAnswer = JudgeBooleanAnswer | JudgeChoiceAnswer | JudgeScoreAnswer;
86
+ /** The answer a question gets, typed by the question: a choice's label is one of its criteria's keys. */
87
+ export type JudgeAnswerFor<Q extends JudgeQuestion> = Q extends JudgeBooleanQuestion ? JudgeBooleanAnswer : Q extends JudgeChoiceQuestion ? JudgeChoiceAnswer<Extract<keyof Q["criteria"], string>> : Q extends JudgeScoreQuestion ? JudgeScoreAnswer : never;
88
+ export type JudgeAnswers<Q extends JudgeQuestions> = {
89
+ [K in keyof Q]: JudgeAnswerFor<Q[K]>;
90
+ };
91
+ /** What a decision comes back as. */
92
+ export type JudgeVerdict<Q extends JudgeQuestions = JudgeQuestions> = {
93
+ /** One answer per question, under the question's id. */
94
+ answers: JudgeAnswers<Q>;
95
+ /** The decision model that answered. */
96
+ model: string;
97
+ /** What the judgement cost, USD. It is already on the account's ledger. */
98
+ costUsd: number;
99
+ };
100
+ /** The route's whole answer. */
101
+ export type JudgeResponse<Q extends JudgeQuestions = JudgeQuestions> = JudgeVerdict<Q> & {
102
+ success: true;
103
+ };
@@ -0,0 +1,12 @@
1
+ "use strict";
2
+ /**
3
+ * The judge: a yes/no, pick-one or score decision made by a decision model, on facts the caller
4
+ * states.
5
+ *
6
+ * Not a conversation. It writes no text and gives no reason: one call, every question judged
7
+ * against the same state, and typed answers back with probabilities. The state is facts (the
8
+ * message, who sent it, the rules, numbers and dates already worked out in code), never a
9
+ * transcript or an argument for an answer. State and the longest question together fit in about
10
+ * 32,000 tokens.
11
+ */
12
+ Object.defineProperty(exports, "__esModule", { value: true });
@@ -39,6 +39,14 @@ export type ToolStatus = {
39
39
  content?: ToolStatusContent[];
40
40
  /** Whether this tool call is ending the conversation */
41
41
  endingConvo?: boolean;
42
+ /** The tool that ran: its id, or a raw (MCP) tool's key. A fact for clients to present, never a label. */
43
+ toolId?: string;
44
+ /** Epoch ms of the first status emitted for this id; the same on every later status for it. */
45
+ startedAt?: number;
46
+ /** Epoch ms at which this id reached `completed` or `failed`. */
47
+ endedAt?: number;
48
+ /** The id of an earlier failed call of the same tool, in the same turn, that this call retries. */
49
+ retryOf?: string;
42
50
  };
43
51
  export type BaseResponseMetadata = {
44
52
  /** Unique response ID */
@@ -7,4 +7,5 @@ export * from "./conversation/address";
7
7
  export * from "./conversation/context";
8
8
  export * from "./jobs";
9
9
  export * from "./outreach";
10
+ export * from "./judge";
10
11
  export * from "./error";
@@ -23,4 +23,5 @@ __exportStar(require("./conversation/address"), exports);
23
23
  __exportStar(require("./conversation/context"), exports);
24
24
  __exportStar(require("./jobs"), exports);
25
25
  __exportStar(require("./outreach"), exports);
26
+ __exportStar(require("./judge"), exports);
26
27
  __exportStar(require("./error"), exports);
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@butlerbot/sdk",
3
- "version": "0.0.46",
3
+ "version": "0.0.47",
4
4
  "description": "The official ButlerBot SDK",
5
5
  "main": "dist/index.js",
6
6
  "types": "dist/index.d.ts",
package/readme.md CHANGED
@@ -387,10 +387,63 @@ one, and answering it is refused with a 409 just as one already answered is. `so
387
387
  wrote a delivery: `shift` for a question a job's shift asked, `runtime` for the job's own
388
388
  status tells, `gate` for an approval the autonomy gate asked for.
389
389
 
390
+ ## Judge
391
+
392
+ A yes/no, pick-one or score decision on facts you state, made by a decision model: well under
393
+ a second and a fraction of a cent, so an app can ask it per event. It is Alfred's own judge,
394
+ the one his jobs and reflexes decide with, for your code to decide with too.
395
+
396
+ ```typescript
397
+ const { answers, model, costUsd } = await client.judge({
398
+ state: { message: text, author: "bot: false", channel: "general", rules },
399
+ questions: {
400
+ hostile: {
401
+ type: "boolean",
402
+ instructions: "Is the message hostile?",
403
+ criteria: { true: "It attacks or insults someone", false: "It is civil, however blunt" },
404
+ },
405
+ rule: {
406
+ type: "choice",
407
+ instructions: "Which server rule does the message break?",
408
+ criteria: { spam: "Promotional or repeated", abuse: "Insults a person", none: "No rule is broken" },
409
+ },
410
+ heat: {
411
+ type: "score",
412
+ instructions: "How heated is the message?",
413
+ criteria: ["Calm", "Annoyed", "Furious"],
414
+ },
415
+ },
416
+ });
417
+
418
+ answers.hostile.value; // boolean: value, probability
419
+ answers.rule.choice; // "spam" | "abuse" | "none", with confidence and every label's probability
420
+ answers.heat.level; // 0..2, with score (the weighted mean) and confidence
421
+ ```
422
+
423
+ Every question is judged against the same `state` in one call, and the answers come back keyed
424
+ by question id and typed by the question: a choice's `choice` is one of its own criteria keys.
425
+ It is a decision model, not a chat model: it writes no text and does no reasoning, so
426
+
427
+ - state facts, never a transcript or an argument for an answer;
428
+ - work out numbers and dates in code and state the result;
429
+ - describe both sides of a boolean, since the model weighs the state against each;
430
+ - add a `none` label to a choice when nothing may fit, or the nearest label is picked however
431
+ poor the fit;
432
+ - keep a score to ten levels, `criteria[0]` the bottom.
433
+
434
+ State and the longest question together fit in about 32,000 tokens. A question the judge cannot
435
+ ask is a 400 carrying its message, and a judgement it could not reach is a 503: both throw
436
+ `ButlerBotAPIError`, never a guess. Needs `judge.run`.
437
+
438
+ The judge is for a decision your app makes itself. For Alfred to react to something, do not
439
+ judge first: report it as a [hook event](#hooks-emit-or-report-what-matched) and the user's
440
+ reflex decides, with the judge inside it when it needs one. That keeps what Alfred reacts to,
441
+ and what it costs, in the user's hands.
442
+
390
443
  ### When a call fails
391
444
 
392
- Every jobs and outreach call throws `ButlerBotAPIError` when the server does not answer with a
393
- success. The status is on the error, so the cases worth branching on are told apart without
445
+ Every jobs, outreach and judge call throws `ButlerBotAPIError` when the server does not answer
446
+ with a success. The status is on the error, so the cases worth branching on are told apart without
394
447
  reading a message, and the parsed body is kept — a rejected cancel still carries the job, a
395
448
  rejected answer still carries the delivery:
396
449