logisheets-logician 1.15.0 → 1.16.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/agent/loop.d.ts +35 -3
- package/dist/agent/loop.js +26 -11
- package/dist/conversation.d.ts +12 -6
- package/dist/craft-ai.d.ts +67 -0
- package/dist/craft-ai.js +214 -0
- package/dist/craft-ai.test.d.ts +1 -0
- package/dist/craft-ai.test.js +201 -0
- package/dist/craft-interactions-api.d.ts +5 -3
- package/dist/craft-interactions-core.d.ts +5 -0
- package/dist/craft-interactions-core.js +5 -0
- package/dist/crafts/manifest.d.ts +29 -0
- package/dist/crafts/skill-tools.d.ts +14 -1
- package/dist/crafts/skill-tools.js +15 -2
- package/dist/crafts/store.d.ts +4 -1
- package/dist/crafts/store.js +3 -0
- package/dist/index.d.ts +45 -0
- package/dist/index.js +45 -0
- package/dist/index.node.js +395 -52
- package/dist/projection.d.ts +14 -0
- package/dist/projection.js +10 -0
- package/dist/storage.d.ts +21 -5
- package/dist/storage.js +6 -3
- package/dist/tool.d.ts +42 -8
- package/dist/tool.js +16 -4
- package/dist/tools/builder.d.ts +23 -9
- package/dist/tools/builder.js +135 -29
- package/dist/tools/cells.js +6 -0
- package/dist/tools/charts.d.ts +2 -0
- package/dist/tools/charts.js +2 -0
- package/dist/tools/edit.d.ts +17 -2
- package/dist/tools/edit.js +5 -1
- package/dist/tools/history.d.ts +5 -0
- package/dist/tools/history.js +5 -0
- package/dist/tools/inspect.d.ts +6 -5
- package/dist/tools/inspect.js +5 -5
- package/dist/tools/names.d.ts +32 -0
- package/dist/tools/names.js +135 -0
- package/dist/tools/names.test.d.ts +1 -0
- package/dist/tools/names.test.js +69 -0
- package/dist/tools/pivot.test.js +59 -1
- package/dist/tools/rename-preserves-schema.test.d.ts +1 -0
- package/dist/tools/rename-preserves-schema.test.js +182 -0
- package/dist/tools/taxonomy.js +2 -0
- package/dist/tools/temp-branch.test.js +20 -0
- package/package.json +3 -3
package/dist/agent/loop.d.ts
CHANGED
|
@@ -50,10 +50,16 @@ export interface LlmResponse {
|
|
|
50
50
|
cache_read_input_tokens?: number;
|
|
51
51
|
};
|
|
52
52
|
}
|
|
53
|
+
/**
|
|
54
|
+
* The one call the loop makes. Implementations translate to their wire format
|
|
55
|
+
* and back; a rejected promise (network, HTTP error, abort) propagates out of
|
|
56
|
+
* `Agent.runTurn` / `askAi` unchanged — neither retries.
|
|
57
|
+
*/
|
|
53
58
|
export interface LlmClient {
|
|
54
59
|
createMessage(params: LlmCreateMessageParams): Promise<LlmResponse>;
|
|
55
60
|
}
|
|
56
61
|
export interface AgentOptions {
|
|
62
|
+
/** Must already hold the conversation `runTurn` is called with. */
|
|
57
63
|
store: ConversationStore;
|
|
58
64
|
registry: ToolRegistry;
|
|
59
65
|
llm: LlmClient;
|
|
@@ -62,11 +68,22 @@ export interface AgentOptions {
|
|
|
62
68
|
model?: string;
|
|
63
69
|
/** Cap on tokens per response. Default 4096. */
|
|
64
70
|
max_tokens?: number;
|
|
65
|
-
/**
|
|
71
|
+
/**
|
|
72
|
+
* System prompt, sent verbatim as the only system block. Tools travel in
|
|
73
|
+
* the request's `tools` field, not in this text.
|
|
74
|
+
*/
|
|
66
75
|
systemPrompt: string;
|
|
67
|
-
/**
|
|
76
|
+
/**
|
|
77
|
+
* Defensive cap on LLM round-trips per user turn (not individual tool
|
|
78
|
+
* calls — one response may hold several). Default 16. The check runs
|
|
79
|
+
* before each request as `iter++ > cap`, so up to cap + 1 requests go out.
|
|
80
|
+
*/
|
|
68
81
|
max_tool_iterations?: number;
|
|
69
|
-
/**
|
|
82
|
+
/**
|
|
83
|
+
* Confirmation prompt for tools whose policy demands it. Omitted means
|
|
84
|
+
* auto-approve everything, which suits tests and trusted headless hosts
|
|
85
|
+
* only. `'once'` is passed through as-is; remembering it is on the host.
|
|
86
|
+
*/
|
|
70
87
|
confirm?: (toolName: string, input: unknown, policy: 'once' | 'always' | 'destructive') => Promise<{
|
|
71
88
|
approved: boolean;
|
|
72
89
|
reason?: string;
|
|
@@ -78,6 +95,11 @@ export interface AgentOptions {
|
|
|
78
95
|
* overlay-widget state. Omit on headless hosts. */
|
|
79
96
|
craftInteractions?: CraftInteractionsApi;
|
|
80
97
|
}
|
|
98
|
+
/**
|
|
99
|
+
* One agent bound to one workbook, store and registry. Reusable across
|
|
100
|
+
* conversations and turns; holds no per-conversation state of its own —
|
|
101
|
+
* everything is re-read from the store before each LLM request.
|
|
102
|
+
*/
|
|
81
103
|
export declare class Agent {
|
|
82
104
|
private store;
|
|
83
105
|
private registry;
|
|
@@ -94,6 +116,16 @@ export declare class Agent {
|
|
|
94
116
|
/**
|
|
95
117
|
* Run one user turn end-to-end: append the user_message, then loop
|
|
96
118
|
* LLM ↔ tools until the model emits end_turn or we hit a safety cap.
|
|
119
|
+
*
|
|
120
|
+
* Requires `conversation_id` to exist in the store (create it first;
|
|
121
|
+
* `MemoryConversationStore.appendEvent` throws otherwise). Results arrive
|
|
122
|
+
* as appended events, not as a return value — subscribe to the store or
|
|
123
|
+
* re-list its events to render them.
|
|
124
|
+
*
|
|
125
|
+
* Failure: tool errors and declined confirmations become `tool_result`
|
|
126
|
+
* events and the loop continues; an `LlmClient` or store rejection
|
|
127
|
+
* propagates and ends the turn. Hitting the cap, max_tokens, or an abort
|
|
128
|
+
* seen at the top of the loop ends it with a `ui_only` system_note.
|
|
97
129
|
*/
|
|
98
130
|
runTurn(conversation_id: string, userText: string, signal?: AbortSignal): Promise<void>;
|
|
99
131
|
private executeToolCall;
|
package/dist/agent/loop.js
CHANGED
|
@@ -18,6 +18,11 @@
|
|
|
18
18
|
* and lets us test it without a network.
|
|
19
19
|
*/
|
|
20
20
|
import { toLlmMessages, } from '../projection.js';
|
|
21
|
+
/**
|
|
22
|
+
* One agent bound to one workbook, store and registry. Reusable across
|
|
23
|
+
* conversations and turns; holds no per-conversation state of its own —
|
|
24
|
+
* everything is re-read from the store before each LLM request.
|
|
25
|
+
*/
|
|
21
26
|
export class Agent {
|
|
22
27
|
constructor(opts) {
|
|
23
28
|
var _a, _b, _c, _d, _e;
|
|
@@ -31,14 +36,23 @@ export class Agent {
|
|
|
31
36
|
this.maxToolIters = (_c = opts.max_tool_iterations) !== null && _c !== void 0 ? _c : 16;
|
|
32
37
|
// Default confirm: auto-approve. Browser host overrides with a
|
|
33
38
|
// real modal. CLI host can override with stdin prompt.
|
|
34
|
-
this.confirm =
|
|
35
|
-
(_d = opts.confirm) !== null && _d !== void 0 ? _d : (async () => ({ approved: true }));
|
|
39
|
+
this.confirm = (_d = opts.confirm) !== null && _d !== void 0 ? _d : (async () => ({ approved: true }));
|
|
36
40
|
this.log = (_e = opts.log) !== null && _e !== void 0 ? _e : (() => { });
|
|
37
41
|
this.craftInteractions = opts.craftInteractions;
|
|
38
42
|
}
|
|
39
43
|
/**
|
|
40
44
|
* Run one user turn end-to-end: append the user_message, then loop
|
|
41
45
|
* LLM ↔ tools until the model emits end_turn or we hit a safety cap.
|
|
46
|
+
*
|
|
47
|
+
* Requires `conversation_id` to exist in the store (create it first;
|
|
48
|
+
* `MemoryConversationStore.appendEvent` throws otherwise). Results arrive
|
|
49
|
+
* as appended events, not as a return value — subscribe to the store or
|
|
50
|
+
* re-list its events to render them.
|
|
51
|
+
*
|
|
52
|
+
* Failure: tool errors and declined confirmations become `tool_result`
|
|
53
|
+
* events and the loop continues; an `LlmClient` or store rejection
|
|
54
|
+
* propagates and ends the turn. Hitting the cap, max_tokens, or an abort
|
|
55
|
+
* seen at the top of the loop ends it with a `ui_only` system_note.
|
|
42
56
|
*/
|
|
43
57
|
async runTurn(conversation_id, userText, signal) {
|
|
44
58
|
const userEvent = {
|
|
@@ -50,9 +64,10 @@ export class Agent {
|
|
|
50
64
|
};
|
|
51
65
|
await this.store.appendEvent(userEvent);
|
|
52
66
|
const blobResolver = (ref) => {
|
|
53
|
-
//
|
|
54
|
-
//
|
|
55
|
-
//
|
|
67
|
+
// Blob refs are NOT resolved yet: `getBlob` is async and nothing
|
|
68
|
+
// pre-hydrates a sync cache, so a `blob_ref` tool output reaches
|
|
69
|
+
// the model as the "(blob … not resolved)" placeholder. The loop
|
|
70
|
+
// itself never writes blob refs; only a host-written event would.
|
|
56
71
|
return null;
|
|
57
72
|
};
|
|
58
73
|
let iter = 0;
|
|
@@ -117,6 +132,8 @@ export class Agent {
|
|
|
117
132
|
// requires *all* tool_results for the previous turn before the
|
|
118
133
|
// next request, so we walk them sequentially here.
|
|
119
134
|
for (const call of toolCalls) {
|
|
135
|
+
// Returning here leaves the remaining tool_use blocks with no
|
|
136
|
+
// tool_result on record.
|
|
120
137
|
if (signal === null || signal === void 0 ? void 0 : signal.aborted)
|
|
121
138
|
return;
|
|
122
139
|
await this.executeToolCall(conversation_id, call, signal);
|
|
@@ -219,12 +236,10 @@ export class Agent {
|
|
|
219
236
|
}
|
|
220
237
|
}
|
|
221
238
|
buildSystem() {
|
|
222
|
-
//
|
|
223
|
-
//
|
|
224
|
-
//
|
|
225
|
-
//
|
|
226
|
-
// The Messages API treats every block as a sub-prompt; the last
|
|
227
|
-
// cache_control marker wins for that prefix.
|
|
239
|
+
// A single cached block holding the host's system prompt. Tools are
|
|
240
|
+
// not repeated here: they go in the request's `tools` field, which
|
|
241
|
+
// precedes `system` in Anthropic's cache prefix, so this breakpoint
|
|
242
|
+
// covers them too.
|
|
228
243
|
return [
|
|
229
244
|
{
|
|
230
245
|
type: 'text',
|
package/dist/conversation.d.ts
CHANGED
|
@@ -37,7 +37,8 @@ export interface ToolCallEvent extends BaseEvent {
|
|
|
37
37
|
tool_use_id: string;
|
|
38
38
|
/** Fully-qualified tool name, e.g. "build__create_block". */
|
|
39
39
|
name: string;
|
|
40
|
-
/**
|
|
40
|
+
/** Input exactly as the model sent it — not validated against the
|
|
41
|
+
* tool's schema. */
|
|
41
42
|
input: unknown;
|
|
42
43
|
turn_id: string;
|
|
43
44
|
}
|
|
@@ -45,13 +46,18 @@ export interface ToolResultEvent extends BaseEvent {
|
|
|
45
46
|
kind: 'tool_result';
|
|
46
47
|
tool_use_id: string;
|
|
47
48
|
/**
|
|
48
|
-
* Output payload, or a blob ref when the value is large.
|
|
49
|
-
*
|
|
50
|
-
* `
|
|
51
|
-
* `
|
|
49
|
+
* Output payload, or a blob ref when the value is large. A host may store
|
|
50
|
+
* a large output via `ConversationStore.putBlob` and reference it here as
|
|
51
|
+
* `{kind: 'blob_ref', ref: '...'}` to keep the events table compact. The
|
|
52
|
+
* `Agent` never does this itself, and it does not resolve refs when
|
|
53
|
+
* projecting (see agent/loop.ts), so the model would see a placeholder.
|
|
52
54
|
*/
|
|
53
55
|
output: unknown;
|
|
54
|
-
/**
|
|
56
|
+
/**
|
|
57
|
+
* Set when the call failed: the handler threw, the tool was unknown, or
|
|
58
|
+
* the user declined. The `Agent` then writes `output: null`, and the
|
|
59
|
+
* projection sends this string as an `is_error` tool_result.
|
|
60
|
+
*/
|
|
55
61
|
error?: string;
|
|
56
62
|
/** Total time the handler spent, ms. */
|
|
57
63
|
duration_ms: number;
|
|
@@ -0,0 +1,67 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `askAi` — the craft→AI direction.
|
|
3
|
+
*
|
|
4
|
+
* The three craft faces all run inward: something else calls the craft. This is
|
|
5
|
+
* the one that runs outward, so a craft can put a question to a model and get a
|
|
6
|
+
* schema-valid answer: a chess board asking for the opponent's move, a
|
|
7
|
+
* simulator asking what to look at next.
|
|
8
|
+
*
|
|
9
|
+
* Two properties make it different from `Agent`:
|
|
10
|
+
*
|
|
11
|
+
* - **Stateless.** Nothing survives the call. A craft's state is queryable
|
|
12
|
+
* from the workbook, so the model re-reads the live position every time
|
|
13
|
+
* rather than replaying a history that can go stale.
|
|
14
|
+
* - **Read-only.** The loop is offered the craft's own `@mutates none` tools
|
|
15
|
+
* and nothing else — no `EDIT_TOOLS`, no workbook surface. The model
|
|
16
|
+
* gathers; the craft decides what to do with the answer. That is what keeps
|
|
17
|
+
* an `askAi` retryable and cancellable: abandoning one mid-loop cannot have
|
|
18
|
+
* half-changed the workbook.
|
|
19
|
+
*/
|
|
20
|
+
import type { LlmClient } from './agent/loop.js';
|
|
21
|
+
import { type JSONSchema, type Tool, type ToolContext } from './tool.js';
|
|
22
|
+
/** One question a craft can ask, as `craftsmith` extracted it from `@aiRole`. */
|
|
23
|
+
export interface AiRole {
|
|
24
|
+
/** Used only in error messages here; the host selects roles by it. */
|
|
25
|
+
name: string;
|
|
26
|
+
/** Sent verbatim as the (cached) system prompt. */
|
|
27
|
+
system: string;
|
|
28
|
+
/**
|
|
29
|
+
* Becomes the `reply` tool's input schema. Only checked shallowly: an
|
|
30
|
+
* object, `required` keys present, top-level primitive `type`s and
|
|
31
|
+
* `enum`s — nested shapes are the craft's to validate.
|
|
32
|
+
*/
|
|
33
|
+
replySchema: JSONSchema;
|
|
34
|
+
}
|
|
35
|
+
export interface AskAiParams {
|
|
36
|
+
llm: LlmClient;
|
|
37
|
+
model: string;
|
|
38
|
+
role: AiRole;
|
|
39
|
+
/** The question. Not the state — that is what the tools are for. */
|
|
40
|
+
input: string;
|
|
41
|
+
/** The craft's read-only tools. A mutating tool here is a caller bug. */
|
|
42
|
+
tools: readonly Tool[];
|
|
43
|
+
/** Context handed to a dispatched craft tool; its `signal` also aborts us. */
|
|
44
|
+
ctx: ToolContext;
|
|
45
|
+
/** Per-response output cap. Default 16384 — see DEFAULT_MAX_TOKENS. */
|
|
46
|
+
max_tokens?: number;
|
|
47
|
+
/** LLM round-trips allowed, the answering one included. Default 8. */
|
|
48
|
+
max_iterations?: number;
|
|
49
|
+
}
|
|
50
|
+
export declare class AskAiError extends Error {
|
|
51
|
+
/** The turn the loop gave up on, for a host that wants to log it. */
|
|
52
|
+
readonly detail?: unknown | undefined;
|
|
53
|
+
constructor(message: string,
|
|
54
|
+
/** The turn the loop gave up on, for a host that wants to log it. */
|
|
55
|
+
detail?: unknown | undefined);
|
|
56
|
+
}
|
|
57
|
+
/**
|
|
58
|
+
* Run one question to completion and return the model's parsed answer.
|
|
59
|
+
*
|
|
60
|
+
* Throws {@link AskAiError} if `tools` contains a mutating tool (before any
|
|
61
|
+
* request), a response stops at max_tokens, the model answers in prose twice,
|
|
62
|
+
* answers in the wrong shape twice, or runs past `max_iterations`. Anything
|
|
63
|
+
* else is not wrapped: an abort of `ctx.signal` throws the signal's reason
|
|
64
|
+
* and an `LlmClient` rejection propagates. A craft tool that throws does not
|
|
65
|
+
* end the call; the model gets the message as an `is_error` result.
|
|
66
|
+
*/
|
|
67
|
+
export declare function askAi<T = unknown>(params: AskAiParams): Promise<T>;
|
package/dist/craft-ai.js
ADDED
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* `askAi` — the craft→AI direction.
|
|
3
|
+
*
|
|
4
|
+
* The three craft faces all run inward: something else calls the craft. This is
|
|
5
|
+
* the one that runs outward, so a craft can put a question to a model and get a
|
|
6
|
+
* schema-valid answer: a chess board asking for the opponent's move, a
|
|
7
|
+
* simulator asking what to look at next.
|
|
8
|
+
*
|
|
9
|
+
* Two properties make it different from `Agent`:
|
|
10
|
+
*
|
|
11
|
+
* - **Stateless.** Nothing survives the call. A craft's state is queryable
|
|
12
|
+
* from the workbook, so the model re-reads the live position every time
|
|
13
|
+
* rather than replaying a history that can go stale.
|
|
14
|
+
* - **Read-only.** The loop is offered the craft's own `@mutates none` tools
|
|
15
|
+
* and nothing else — no `EDIT_TOOLS`, no workbook surface. The model
|
|
16
|
+
* gathers; the craft decides what to do with the answer. That is what keeps
|
|
17
|
+
* an `askAi` retryable and cancellable: abandoning one mid-loop cannot have
|
|
18
|
+
* half-changed the workbook.
|
|
19
|
+
*/
|
|
20
|
+
import { toLlmTool, } from './tool.js';
|
|
21
|
+
/** The reply tool's id. Not namespaced — it is synthetic, not a craft tool. */
|
|
22
|
+
const REPLY = 'reply';
|
|
23
|
+
/** Safety and cost ceiling: how many times the model may go back for state. */
|
|
24
|
+
const DEFAULT_MAX_ITERATIONS = 8;
|
|
25
|
+
/**
|
|
26
|
+
* Room for one turn's output.
|
|
27
|
+
*
|
|
28
|
+
* Generous because a reasoning model spends nearly all of it on a `thinking`
|
|
29
|
+
* block before it emits anything. Measured against claude-opus-5 deciding a
|
|
30
|
+
* single 四象 draft pick: 1024 was cut off mid-thought every time, and 4096
|
|
31
|
+
* still was. The role's own prompt is what drives this — it says to work the
|
|
32
|
+
* position out rather than be handed a score — so the budget has to cover
|
|
33
|
+
* reasoning, not the answer, which is a few dozen tokens.
|
|
34
|
+
*
|
|
35
|
+
* A role that wants a tighter leash passes its own `max_tokens`.
|
|
36
|
+
*/
|
|
37
|
+
const DEFAULT_MAX_TOKENS = 16384;
|
|
38
|
+
export class AskAiError extends Error {
|
|
39
|
+
constructor(message,
|
|
40
|
+
/** The turn the loop gave up on, for a host that wants to log it. */
|
|
41
|
+
detail) {
|
|
42
|
+
super(message);
|
|
43
|
+
this.detail = detail;
|
|
44
|
+
this.name = 'AskAiError';
|
|
45
|
+
}
|
|
46
|
+
}
|
|
47
|
+
/**
|
|
48
|
+
* The reply schema as one more tool in the set. It cannot be forced with
|
|
49
|
+
* `tool_choice` — the model has to be free to read first — so termination is
|
|
50
|
+
* this loop's job: it ends when the model calls `reply`.
|
|
51
|
+
*/
|
|
52
|
+
function replyTool(role) {
|
|
53
|
+
return {
|
|
54
|
+
name: REPLY,
|
|
55
|
+
description: 'Give your final answer. Call this exactly once, when you have ' +
|
|
56
|
+
'read everything you need.',
|
|
57
|
+
input_schema: { type: 'object', ...role.replySchema },
|
|
58
|
+
};
|
|
59
|
+
}
|
|
60
|
+
/**
|
|
61
|
+
* Shallow structural check against the reply schema — enough to catch a model
|
|
62
|
+
* that answered with the wrong shape, which is the failure this guards. Deep
|
|
63
|
+
* validation is the craft's business; it knows what a legal move is.
|
|
64
|
+
*/
|
|
65
|
+
function schemaViolations(schema, value) {
|
|
66
|
+
var _a, _b;
|
|
67
|
+
if (typeof value !== 'object' || value === null || Array.isArray(value))
|
|
68
|
+
return ['the answer must be an object'];
|
|
69
|
+
const obj = value;
|
|
70
|
+
const out = [];
|
|
71
|
+
for (const key of (_a = schema.required) !== null && _a !== void 0 ? _a : [])
|
|
72
|
+
if (obj[key] === undefined)
|
|
73
|
+
out.push(`"${key}" is required but missing`);
|
|
74
|
+
for (const [key, spec] of Object.entries((_b = schema.properties) !== null && _b !== void 0 ? _b : {})) {
|
|
75
|
+
const v = obj[key];
|
|
76
|
+
if (v === undefined)
|
|
77
|
+
continue;
|
|
78
|
+
const want = spec.type;
|
|
79
|
+
const got = Array.isArray(v) ? 'array' : typeof v;
|
|
80
|
+
if (want === 'number' && got !== 'number')
|
|
81
|
+
out.push(`"${key}" must be a number, got ${got}`);
|
|
82
|
+
else if (want === 'string' && got !== 'string')
|
|
83
|
+
out.push(`"${key}" must be a string, got ${got}`);
|
|
84
|
+
else if (want === 'boolean' && got !== 'boolean')
|
|
85
|
+
out.push(`"${key}" must be a boolean, got ${got}`);
|
|
86
|
+
else if (want === 'array' && got !== 'array')
|
|
87
|
+
out.push(`"${key}" must be an array, got ${got}`);
|
|
88
|
+
const allowed = spec.enum;
|
|
89
|
+
if (allowed && !allowed.includes(v))
|
|
90
|
+
out.push(`"${key}" must be one of ${allowed.join(', ')}`);
|
|
91
|
+
}
|
|
92
|
+
return out;
|
|
93
|
+
}
|
|
94
|
+
function textOf(content) {
|
|
95
|
+
return content
|
|
96
|
+
.filter((b) => b.type === 'text')
|
|
97
|
+
.map((b) => b.text)
|
|
98
|
+
.join(' ')
|
|
99
|
+
.trim();
|
|
100
|
+
}
|
|
101
|
+
/**
|
|
102
|
+
* Run one question to completion and return the model's parsed answer.
|
|
103
|
+
*
|
|
104
|
+
* Throws {@link AskAiError} if `tools` contains a mutating tool (before any
|
|
105
|
+
* request), a response stops at max_tokens, the model answers in prose twice,
|
|
106
|
+
* answers in the wrong shape twice, or runs past `max_iterations`. Anything
|
|
107
|
+
* else is not wrapped: an abort of `ctx.signal` throws the signal's reason
|
|
108
|
+
* and an `LlmClient` rejection propagates. A craft tool that throws does not
|
|
109
|
+
* end the call; the model gets the message as an `is_error` result.
|
|
110
|
+
*/
|
|
111
|
+
export async function askAi(params) {
|
|
112
|
+
var _a, _b, _c, _d;
|
|
113
|
+
const { llm, model, role, input, tools, ctx } = params;
|
|
114
|
+
const maxIterations = (_a = params.max_iterations) !== null && _a !== void 0 ? _a : DEFAULT_MAX_ITERATIONS;
|
|
115
|
+
const mutating = tools.filter((t) => t.mutates);
|
|
116
|
+
if (mutating.length)
|
|
117
|
+
throw new AskAiError(`role "${role.name}" was given mutating tool(s) ` +
|
|
118
|
+
`${mutating
|
|
119
|
+
.map((t) => t.name)
|
|
120
|
+
.join(', ')}; an askAi loop reads only`);
|
|
121
|
+
const byId = new Map(tools.map((t) => [`${t.namespace}__${t.name}`, t]));
|
|
122
|
+
const llmTools = [...tools.map(toLlmTool), replyTool(role)];
|
|
123
|
+
const system = [
|
|
124
|
+
{ type: 'text', text: role.system, cache_control: { type: 'ephemeral' } },
|
|
125
|
+
];
|
|
126
|
+
const messages = [{ role: 'user', content: input }];
|
|
127
|
+
let retriedShape = false;
|
|
128
|
+
let noAnswerNudges = 0;
|
|
129
|
+
for (let i = 0; i < maxIterations; i++) {
|
|
130
|
+
ctx.signal.throwIfAborted();
|
|
131
|
+
const res = await llm.createMessage({
|
|
132
|
+
model,
|
|
133
|
+
system,
|
|
134
|
+
tools: llmTools,
|
|
135
|
+
messages,
|
|
136
|
+
max_tokens: (_b = params.max_tokens) !== null && _b !== void 0 ? _b : DEFAULT_MAX_TOKENS,
|
|
137
|
+
signal: ctx.signal,
|
|
138
|
+
});
|
|
139
|
+
messages.push({ role: 'assistant', content: res.content });
|
|
140
|
+
// A truncated turn can carry no usable tool call, so left alone it
|
|
141
|
+
// silently spends an iteration and the loop ends up reporting that the
|
|
142
|
+
// model never answered — which is not what went wrong.
|
|
143
|
+
if (res.stop_reason === 'max_tokens')
|
|
144
|
+
throw new AskAiError(`role "${role.name}" was cut off at max_tokens ` +
|
|
145
|
+
`(${(_c = params.max_tokens) !== null && _c !== void 0 ? _c : DEFAULT_MAX_TOKENS}) before it ` +
|
|
146
|
+
`could answer; raise max_tokens for this role`, textOf(res.content));
|
|
147
|
+
const calls = res.content.filter((b) => b.type === 'tool_use');
|
|
148
|
+
const answer = calls.find((c) => c.name === REPLY);
|
|
149
|
+
if (answer) {
|
|
150
|
+
const bad = schemaViolations(role.replySchema, answer.input);
|
|
151
|
+
if (!bad.length)
|
|
152
|
+
return answer.input;
|
|
153
|
+
if (retriedShape)
|
|
154
|
+
throw new AskAiError(`role "${role.name}" answered in the wrong shape: ${bad.join('; ')}`, answer.input);
|
|
155
|
+
retriedShape = true;
|
|
156
|
+
messages.push({
|
|
157
|
+
role: 'user',
|
|
158
|
+
content: [
|
|
159
|
+
{
|
|
160
|
+
type: 'tool_result',
|
|
161
|
+
tool_use_id: answer.id,
|
|
162
|
+
content: `That answer does not fit: ${bad.join('; ')}. Call ${REPLY} again, corrected.`,
|
|
163
|
+
is_error: true,
|
|
164
|
+
},
|
|
165
|
+
],
|
|
166
|
+
});
|
|
167
|
+
continue;
|
|
168
|
+
}
|
|
169
|
+
if (!calls.length) {
|
|
170
|
+
// A turn of prose instead of an answer. Nudge once; a model that
|
|
171
|
+
// still will not use the tool is not going to.
|
|
172
|
+
if (noAnswerNudges++ > 0)
|
|
173
|
+
throw new AskAiError(`role "${role.name}" never called ${REPLY}`, textOf(res.content));
|
|
174
|
+
messages.push({
|
|
175
|
+
role: 'user',
|
|
176
|
+
content: `Answer by calling the ${REPLY} tool.`,
|
|
177
|
+
});
|
|
178
|
+
continue;
|
|
179
|
+
}
|
|
180
|
+
const results = [];
|
|
181
|
+
for (const call of calls) {
|
|
182
|
+
const tool = byId.get(call.name);
|
|
183
|
+
if (!tool) {
|
|
184
|
+
results.push({
|
|
185
|
+
type: 'tool_result',
|
|
186
|
+
tool_use_id: call.id,
|
|
187
|
+
content: `No tool named ${call.name}.`,
|
|
188
|
+
is_error: true,
|
|
189
|
+
});
|
|
190
|
+
continue;
|
|
191
|
+
}
|
|
192
|
+
try {
|
|
193
|
+
const out = await tool.handler(call.input, ctx);
|
|
194
|
+
results.push({
|
|
195
|
+
type: 'tool_result',
|
|
196
|
+
tool_use_id: call.id,
|
|
197
|
+
content: JSON.stringify((_d = out.data) !== null && _d !== void 0 ? _d : null),
|
|
198
|
+
});
|
|
199
|
+
}
|
|
200
|
+
catch (e) {
|
|
201
|
+
// The model gets the reason and can try a different read; only
|
|
202
|
+
// the loop's own limits end the call.
|
|
203
|
+
results.push({
|
|
204
|
+
type: 'tool_result',
|
|
205
|
+
tool_use_id: call.id,
|
|
206
|
+
content: e instanceof Error ? e.message : String(e),
|
|
207
|
+
is_error: true,
|
|
208
|
+
});
|
|
209
|
+
}
|
|
210
|
+
}
|
|
211
|
+
messages.push({ role: 'user', content: results });
|
|
212
|
+
}
|
|
213
|
+
throw new AskAiError(`role "${role.name}" did not answer within ${maxIterations} steps`);
|
|
214
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
export {};
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
import { describe, it, expect, vi } from 'vitest';
|
|
2
|
+
import { askAi, AskAiError } from './craft-ai.js';
|
|
3
|
+
const ROLE = {
|
|
4
|
+
name: 'opponent',
|
|
5
|
+
system: 'You play chess as Black.',
|
|
6
|
+
replySchema: {
|
|
7
|
+
type: 'object',
|
|
8
|
+
properties: { uci: { type: 'string' }, plies: { type: 'number' } },
|
|
9
|
+
required: ['uci'],
|
|
10
|
+
},
|
|
11
|
+
};
|
|
12
|
+
/**
|
|
13
|
+
* Replays a canned script of model turns, snapshotting each request. The loop
|
|
14
|
+
* appends to one `messages` array as it goes, so recording the reference would
|
|
15
|
+
* make every turn look like the last one.
|
|
16
|
+
*/
|
|
17
|
+
function fakeLlm(turns) {
|
|
18
|
+
const sent = [];
|
|
19
|
+
let i = 0;
|
|
20
|
+
return {
|
|
21
|
+
sent,
|
|
22
|
+
async createMessage(params) {
|
|
23
|
+
sent.push({
|
|
24
|
+
system: structuredClone(params.system),
|
|
25
|
+
tools: params.tools.map((t) => ({ name: t.name })),
|
|
26
|
+
messages: structuredClone(params.messages),
|
|
27
|
+
});
|
|
28
|
+
const content = turns[i++];
|
|
29
|
+
if (!content)
|
|
30
|
+
throw new Error('fake llm ran out of turns');
|
|
31
|
+
return {
|
|
32
|
+
content,
|
|
33
|
+
stop_reason: content.some((b) => b.type === 'tool_use')
|
|
34
|
+
? 'tool_use'
|
|
35
|
+
: 'end_turn',
|
|
36
|
+
};
|
|
37
|
+
},
|
|
38
|
+
};
|
|
39
|
+
}
|
|
40
|
+
const use = (name, input = {}, id = name) => ({
|
|
41
|
+
type: 'tool_use',
|
|
42
|
+
id,
|
|
43
|
+
name,
|
|
44
|
+
input,
|
|
45
|
+
});
|
|
46
|
+
const say = (text) => ({ type: 'text', text });
|
|
47
|
+
function readTool(name, data, mutates = false) {
|
|
48
|
+
return {
|
|
49
|
+
namespace: 'chess',
|
|
50
|
+
name,
|
|
51
|
+
description: name,
|
|
52
|
+
inputSchema: { type: 'object', properties: {} },
|
|
53
|
+
mutates,
|
|
54
|
+
handler: vi.fn(async () => ({ data })),
|
|
55
|
+
};
|
|
56
|
+
}
|
|
57
|
+
function ctx(signal = new AbortController().signal) {
|
|
58
|
+
return {
|
|
59
|
+
workbook: {},
|
|
60
|
+
signal,
|
|
61
|
+
confirm: async () => true,
|
|
62
|
+
log: () => { },
|
|
63
|
+
};
|
|
64
|
+
}
|
|
65
|
+
const run = (llm, tools = [], over) => askAi({
|
|
66
|
+
llm,
|
|
67
|
+
model: 'test',
|
|
68
|
+
role: ROLE,
|
|
69
|
+
input: 'Your move.',
|
|
70
|
+
tools,
|
|
71
|
+
ctx: ctx(),
|
|
72
|
+
...over,
|
|
73
|
+
});
|
|
74
|
+
describe('askAi', () => {
|
|
75
|
+
it('returns the reply the model gives', async () => {
|
|
76
|
+
const llm = fakeLlm([[use('reply', { uci: 'g8f6' })]]);
|
|
77
|
+
await expect(run(llm)).resolves.toEqual({ uci: 'g8f6' });
|
|
78
|
+
});
|
|
79
|
+
it('lets the model read through the craft tools before answering', async () => {
|
|
80
|
+
const position = readTool('get_position', { fen: 'startpos' });
|
|
81
|
+
const legal = readTool('get_legal_moves', { moves: ['g8f6'] });
|
|
82
|
+
const llm = fakeLlm([
|
|
83
|
+
[use('chess__get_position')],
|
|
84
|
+
[use('chess__get_legal_moves')],
|
|
85
|
+
[use('reply', { uci: 'g8f6' })],
|
|
86
|
+
]);
|
|
87
|
+
await expect(run(llm, [position, legal])).resolves.toEqual({
|
|
88
|
+
uci: 'g8f6',
|
|
89
|
+
});
|
|
90
|
+
expect(position.handler).toHaveBeenCalledOnce();
|
|
91
|
+
expect(legal.handler).toHaveBeenCalledOnce();
|
|
92
|
+
// The tool result has to come back as the model's next input.
|
|
93
|
+
expect(JSON.stringify(llm.sent[1].messages)).toContain('startpos');
|
|
94
|
+
});
|
|
95
|
+
it('offers the craft tools plus a reply tool, and nothing else', async () => {
|
|
96
|
+
const llm = fakeLlm([[use('reply', { uci: 'g8f6' })]]);
|
|
97
|
+
await run(llm, [readTool('get_position', {})]);
|
|
98
|
+
expect(llm.sent[0].tools.map((t) => t.name)).toEqual([
|
|
99
|
+
'chess__get_position',
|
|
100
|
+
'reply',
|
|
101
|
+
]);
|
|
102
|
+
});
|
|
103
|
+
it('refuses a mutating tool rather than letting the model write', async () => {
|
|
104
|
+
const llm = fakeLlm([[use('reply', { uci: 'g8f6' })]]);
|
|
105
|
+
await expect(run(llm, [readTool('apply_move', {}, true)])).rejects.toThrow(/reads only/);
|
|
106
|
+
});
|
|
107
|
+
it('re-asks once when the reply has the wrong shape', async () => {
|
|
108
|
+
const llm = fakeLlm([
|
|
109
|
+
[use('reply', { plies: 3 })], // missing the required uci
|
|
110
|
+
[use('reply', { uci: 'g8f6' })],
|
|
111
|
+
]);
|
|
112
|
+
await expect(run(llm)).resolves.toEqual({ uci: 'g8f6' });
|
|
113
|
+
expect(JSON.stringify(llm.sent[1].messages)).toContain('is required but missing');
|
|
114
|
+
});
|
|
115
|
+
it('gives up after a second wrong shape, saying what was wrong', async () => {
|
|
116
|
+
const llm = fakeLlm([
|
|
117
|
+
[use('reply', { uci: 1 })],
|
|
118
|
+
[use('reply', { uci: 2 })],
|
|
119
|
+
]);
|
|
120
|
+
await expect(run(llm)).rejects.toThrow(/"uci" must be a string, got number/);
|
|
121
|
+
});
|
|
122
|
+
it('nudges a model that answers in prose, then gives up', async () => {
|
|
123
|
+
const llm = fakeLlm([[say("I'd play Nf6")], [say('Nf6, definitely')]]);
|
|
124
|
+
await expect(run(llm)).rejects.toThrow(/never called reply/);
|
|
125
|
+
expect(JSON.stringify(llm.sent[1].messages)).toContain('Answer by calling the reply tool');
|
|
126
|
+
});
|
|
127
|
+
it('hands a failing tool back to the model instead of ending the call', async () => {
|
|
128
|
+
const boom = {
|
|
129
|
+
...readTool('get_position', null),
|
|
130
|
+
handler: async () => {
|
|
131
|
+
throw new Error('board not initialized');
|
|
132
|
+
},
|
|
133
|
+
};
|
|
134
|
+
const llm = fakeLlm([
|
|
135
|
+
[use('chess__get_position')],
|
|
136
|
+
[use('reply', { uci: 'g8f6' })],
|
|
137
|
+
]);
|
|
138
|
+
await expect(run(llm, [boom])).resolves.toEqual({ uci: 'g8f6' });
|
|
139
|
+
expect(JSON.stringify(llm.sent[1].messages)).toContain('board not initialized');
|
|
140
|
+
});
|
|
141
|
+
it('tells the model when it invents a tool', async () => {
|
|
142
|
+
const llm = fakeLlm([
|
|
143
|
+
[use('chess__get_evaluation')],
|
|
144
|
+
[use('reply', { uci: 'g8f6' })],
|
|
145
|
+
]);
|
|
146
|
+
await expect(run(llm, [readTool('get_position', {})])).resolves.toEqual({
|
|
147
|
+
uci: 'g8f6',
|
|
148
|
+
});
|
|
149
|
+
expect(JSON.stringify(llm.sent[1].messages)).toContain('No tool named chess__get_evaluation');
|
|
150
|
+
});
|
|
151
|
+
it('stops at max_iterations rather than looping on reads', async () => {
|
|
152
|
+
const llm = fakeLlm(Array.from({ length: 10 }, () => [use('chess__get_position')]));
|
|
153
|
+
await expect(run(llm, [readTool('get_position', {})], { max_iterations: 3 })).rejects.toThrow(/did not answer within 3 steps/);
|
|
154
|
+
expect(llm.sent).toHaveLength(3);
|
|
155
|
+
});
|
|
156
|
+
it('aborts when the craft cancels', async () => {
|
|
157
|
+
const ac = new AbortController();
|
|
158
|
+
ac.abort();
|
|
159
|
+
const llm = fakeLlm([[use('reply', { uci: 'g8f6' })]]);
|
|
160
|
+
await expect(askAi({
|
|
161
|
+
llm,
|
|
162
|
+
model: 'test',
|
|
163
|
+
role: ROLE,
|
|
164
|
+
input: 'Your move.',
|
|
165
|
+
tools: [],
|
|
166
|
+
ctx: ctx(ac.signal),
|
|
167
|
+
})).rejects.toThrow();
|
|
168
|
+
expect(llm.sent).toHaveLength(0);
|
|
169
|
+
});
|
|
170
|
+
it('sends the role system prompt as a cacheable block', async () => {
|
|
171
|
+
const llm = fakeLlm([[use('reply', { uci: 'g8f6' })]]);
|
|
172
|
+
await run(llm);
|
|
173
|
+
expect(llm.sent[0].system).toEqual([
|
|
174
|
+
{
|
|
175
|
+
type: 'text',
|
|
176
|
+
text: 'You play chess as Black.',
|
|
177
|
+
cache_control: { type: 'ephemeral' },
|
|
178
|
+
},
|
|
179
|
+
]);
|
|
180
|
+
});
|
|
181
|
+
it('carries no state between calls', async () => {
|
|
182
|
+
const first = fakeLlm([[use('reply', { uci: 'g8f6' })]]);
|
|
183
|
+
const second = fakeLlm([[use('reply', { uci: 'b8c6' })]]);
|
|
184
|
+
await run(first);
|
|
185
|
+
await run(second);
|
|
186
|
+
// The second call starts from the question alone.
|
|
187
|
+
expect(second.sent[0].messages).toEqual([
|
|
188
|
+
{ role: 'user', content: 'Your move.' },
|
|
189
|
+
]);
|
|
190
|
+
});
|
|
191
|
+
});
|
|
192
|
+
describe('AskAiError', () => {
|
|
193
|
+
it('carries the turn it gave up on', async () => {
|
|
194
|
+
const llm = fakeLlm([[say('nope')], [say('still nope')]]);
|
|
195
|
+
await run(llm).catch((e) => {
|
|
196
|
+
expect(e).toBeInstanceOf(AskAiError);
|
|
197
|
+
expect(e.detail).toBe('still nope');
|
|
198
|
+
});
|
|
199
|
+
expect.assertions(2);
|
|
200
|
+
});
|
|
201
|
+
});
|