logisheets-logician 1.15.0 → 1.16.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (45) hide show
  1. package/dist/agent/loop.d.ts +35 -3
  2. package/dist/agent/loop.js +26 -11
  3. package/dist/conversation.d.ts +12 -6
  4. package/dist/craft-ai.d.ts +67 -0
  5. package/dist/craft-ai.js +214 -0
  6. package/dist/craft-ai.test.d.ts +1 -0
  7. package/dist/craft-ai.test.js +201 -0
  8. package/dist/craft-interactions-api.d.ts +5 -3
  9. package/dist/craft-interactions-core.d.ts +5 -0
  10. package/dist/craft-interactions-core.js +5 -0
  11. package/dist/crafts/manifest.d.ts +29 -0
  12. package/dist/crafts/skill-tools.d.ts +14 -1
  13. package/dist/crafts/skill-tools.js +15 -2
  14. package/dist/crafts/store.d.ts +4 -1
  15. package/dist/crafts/store.js +3 -0
  16. package/dist/index.d.ts +45 -0
  17. package/dist/index.js +45 -0
  18. package/dist/index.node.js +395 -52
  19. package/dist/projection.d.ts +14 -0
  20. package/dist/projection.js +10 -0
  21. package/dist/storage.d.ts +21 -5
  22. package/dist/storage.js +6 -3
  23. package/dist/tool.d.ts +42 -8
  24. package/dist/tool.js +16 -4
  25. package/dist/tools/builder.d.ts +23 -9
  26. package/dist/tools/builder.js +135 -29
  27. package/dist/tools/cells.js +6 -0
  28. package/dist/tools/charts.d.ts +2 -0
  29. package/dist/tools/charts.js +2 -0
  30. package/dist/tools/edit.d.ts +17 -2
  31. package/dist/tools/edit.js +5 -1
  32. package/dist/tools/history.d.ts +5 -0
  33. package/dist/tools/history.js +5 -0
  34. package/dist/tools/inspect.d.ts +6 -5
  35. package/dist/tools/inspect.js +5 -5
  36. package/dist/tools/names.d.ts +32 -0
  37. package/dist/tools/names.js +135 -0
  38. package/dist/tools/names.test.d.ts +1 -0
  39. package/dist/tools/names.test.js +69 -0
  40. package/dist/tools/pivot.test.js +59 -1
  41. package/dist/tools/rename-preserves-schema.test.d.ts +1 -0
  42. package/dist/tools/rename-preserves-schema.test.js +182 -0
  43. package/dist/tools/taxonomy.js +2 -0
  44. package/dist/tools/temp-branch.test.js +20 -0
  45. package/package.json +3 -3
@@ -50,10 +50,16 @@ export interface LlmResponse {
50
50
  cache_read_input_tokens?: number;
51
51
  };
52
52
  }
53
+ /**
54
+ * The one call the loop makes. Implementations translate to their wire format
55
+ * and back; a rejected promise (network, HTTP error, abort) propagates out of
56
+ * `Agent.runTurn` / `askAi` unchanged — neither retries.
57
+ */
53
58
  export interface LlmClient {
54
59
  createMessage(params: LlmCreateMessageParams): Promise<LlmResponse>;
55
60
  }
56
61
  export interface AgentOptions {
62
+ /** Must already hold the conversation `runTurn` is called with. */
57
63
  store: ConversationStore;
58
64
  registry: ToolRegistry;
59
65
  llm: LlmClient;
@@ -62,11 +68,22 @@ export interface AgentOptions {
62
68
  model?: string;
63
69
  /** Cap on tokens per response. Default 4096. */
64
70
  max_tokens?: number;
65
- /** Base system prompt (instructions). Tool listing is auto-appended. */
71
+ /**
72
+ * System prompt, sent verbatim as the only system block. Tools travel in
73
+ * the request's `tools` field, not in this text.
74
+ */
66
75
  systemPrompt: string;
67
- /** Defensive cap on tool calls per user turn to prevent runaway loops. */
76
+ /**
77
+ * Defensive cap on LLM round-trips per user turn (not individual tool
78
+ * calls — one response may hold several). Default 16. The check runs
79
+ * before each request as `iter++ > cap`, so up to cap + 1 requests go out.
80
+ */
68
81
  max_tool_iterations?: number;
69
- /** Confirmation prompt for tools whose policy demands it. */
82
+ /**
83
+ * Confirmation prompt for tools whose policy demands it. Omitted means
84
+ * auto-approve everything, which suits tests and trusted headless hosts
85
+ * only. `'once'` is passed through as-is; remembering it is on the host.
86
+ */
70
87
  confirm?: (toolName: string, input: unknown, policy: 'once' | 'always' | 'destructive') => Promise<{
71
88
  approved: boolean;
72
89
  reason?: string;
@@ -78,6 +95,11 @@ export interface AgentOptions {
78
95
  * overlay-widget state. Omit on headless hosts. */
79
96
  craftInteractions?: CraftInteractionsApi;
80
97
  }
98
+ /**
99
+ * One agent bound to one workbook, store and registry. Reusable across
100
+ * conversations and turns; holds no per-conversation state of its own —
101
+ * everything is re-read from the store before each LLM request.
102
+ */
81
103
  export declare class Agent {
82
104
  private store;
83
105
  private registry;
@@ -94,6 +116,16 @@ export declare class Agent {
94
116
  /**
95
117
  * Run one user turn end-to-end: append the user_message, then loop
96
118
  * LLM ↔ tools until the model emits end_turn or we hit a safety cap.
119
+ *
120
+ * Requires `conversation_id` to exist in the store (create it first;
121
+ * `MemoryConversationStore.appendEvent` throws otherwise). Results arrive
122
+ * as appended events, not as a return value — subscribe to the store or
123
+ * re-list its events to render them.
124
+ *
125
+ * Failure: tool errors and declined confirmations become `tool_result`
126
+ * events and the loop continues; an `LlmClient` or store rejection
127
+ * propagates and ends the turn. Hitting the cap, max_tokens, or an abort
128
+ * seen at the top of the loop ends it with a `ui_only` system_note.
97
129
  */
98
130
  runTurn(conversation_id: string, userText: string, signal?: AbortSignal): Promise<void>;
99
131
  private executeToolCall;
@@ -18,6 +18,11 @@
18
18
  * and lets us test it without a network.
19
19
  */
20
20
  import { toLlmMessages, } from '../projection.js';
21
+ /**
22
+ * One agent bound to one workbook, store and registry. Reusable across
23
+ * conversations and turns; holds no per-conversation state of its own —
24
+ * everything is re-read from the store before each LLM request.
25
+ */
21
26
  export class Agent {
22
27
  constructor(opts) {
23
28
  var _a, _b, _c, _d, _e;
@@ -31,14 +36,23 @@ export class Agent {
31
36
  this.maxToolIters = (_c = opts.max_tool_iterations) !== null && _c !== void 0 ? _c : 16;
32
37
  // Default confirm: auto-approve. Browser host overrides with a
33
38
  // real modal. CLI host can override with stdin prompt.
34
- this.confirm =
35
- (_d = opts.confirm) !== null && _d !== void 0 ? _d : (async () => ({ approved: true }));
39
+ this.confirm = (_d = opts.confirm) !== null && _d !== void 0 ? _d : (async () => ({ approved: true }));
36
40
  this.log = (_e = opts.log) !== null && _e !== void 0 ? _e : (() => { });
37
41
  this.craftInteractions = opts.craftInteractions;
38
42
  }
39
43
  /**
40
44
  * Run one user turn end-to-end: append the user_message, then loop
41
45
  * LLM ↔ tools until the model emits end_turn or we hit a safety cap.
46
+ *
47
+ * Requires `conversation_id` to exist in the store (create it first;
48
+ * `MemoryConversationStore.appendEvent` throws otherwise). Results arrive
49
+ * as appended events, not as a return value — subscribe to the store or
50
+ * re-list its events to render them.
51
+ *
52
+ * Failure: tool errors and declined confirmations become `tool_result`
53
+ * events and the loop continues; an `LlmClient` or store rejection
54
+ * propagates and ends the turn. Hitting the cap, max_tokens, or an abort
55
+ * seen at the top of the loop ends it with a `ui_only` system_note.
42
56
  */
43
57
  async runTurn(conversation_id, userText, signal) {
44
58
  const userEvent = {
@@ -50,9 +64,10 @@ export class Agent {
50
64
  };
51
65
  await this.store.appendEvent(userEvent);
52
66
  const blobResolver = (ref) => {
53
- // listEvents is async; the loop pre-hydrates blobs into a
54
- // synchronous cache before each LLM call below. Default
55
- // fallback returns null (projection emits a placeholder).
67
+ // Blob refs are NOT resolved yet: `getBlob` is async and nothing
68
+ // pre-hydrates a sync cache, so a `blob_ref` tool output reaches
69
+ // the model as the "(blob … not resolved)" placeholder. The loop
70
+ // itself never writes blob refs; only a host-written event would.
56
71
  return null;
57
72
  };
58
73
  let iter = 0;
@@ -117,6 +132,8 @@ export class Agent {
117
132
  // requires *all* tool_results for the previous turn before the
118
133
  // next request, so we walk them sequentially here.
119
134
  for (const call of toolCalls) {
135
+ // Returning here leaves the remaining tool_use blocks with no
136
+ // tool_result on record.
120
137
  if (signal === null || signal === void 0 ? void 0 : signal.aborted)
121
138
  return;
122
139
  await this.executeToolCall(conversation_id, call, signal);
@@ -219,12 +236,10 @@ export class Agent {
219
236
  }
220
237
  }
221
238
  buildSystem() {
222
- // Two blocks so prompt caching works cleanly:
223
- // [0] = user-supplied system prompt (stable, cached)
224
- // [1] = tool list (also stable across a turn; we cache here too
225
- // because the same tools are reused across many turns)
226
- // The Messages API treats every block as a sub-prompt; the last
227
- // cache_control marker wins for that prefix.
239
+ // A single cached block holding the host's system prompt. Tools are
240
+ // not repeated here: they go in the request's `tools` field, which
241
+ // precedes `system` in Anthropic's cache prefix, so this breakpoint
242
+ // covers them too.
228
243
  return [
229
244
  {
230
245
  type: 'text',
@@ -37,7 +37,8 @@ export interface ToolCallEvent extends BaseEvent {
37
37
  tool_use_id: string;
38
38
  /** Fully-qualified tool name, e.g. "build__create_block". */
39
39
  name: string;
40
- /** Validated input object. */
40
+ /** Input exactly as the model sent it — not validated against the
41
+ * tool's schema. */
41
42
  input: unknown;
42
43
  turn_id: string;
43
44
  }
@@ -45,13 +46,18 @@ export interface ToolResultEvent extends BaseEvent {
45
46
  kind: 'tool_result';
46
47
  tool_use_id: string;
47
48
  /**
48
- * Output payload, or a blob ref when the value is large. Large outputs
49
- * (e.g. describe_block with include_rows=true) should be stored via
50
- * `ConversationStore.putBlob` and referenced here as
51
- * `{kind: 'blob_ref', ref: '...'}` to keep events table compact.
49
+ * Output payload, or a blob ref when the value is large. A host may store
50
+ * a large output via `ConversationStore.putBlob` and reference it here as
51
+ * `{kind: 'blob_ref', ref: '...'}` to keep the events table compact. The
52
+ * `Agent` never does this itself, and it does not resolve refs when
53
+ * projecting (see agent/loop.ts), so the model would see a placeholder.
52
54
  */
53
55
  output: unknown;
54
- /** Set when the handler threw. `output` then holds an error summary. */
56
+ /**
57
+ * Set when the call failed: the handler threw, the tool was unknown, or
58
+ * the user declined. The `Agent` then writes `output: null`, and the
59
+ * projection sends this string as an `is_error` tool_result.
60
+ */
55
61
  error?: string;
56
62
  /** Total time the handler spent, ms. */
57
63
  duration_ms: number;
@@ -0,0 +1,67 @@
1
+ /**
2
+ * `askAi` — the craft→AI direction.
3
+ *
4
+ * The three craft faces all run inward: something else calls the craft. This is
5
+ * the one that runs outward, so a craft can put a question to a model and get a
6
+ * schema-valid answer: a chess board asking for the opponent's move, a
7
+ * simulator asking what to look at next.
8
+ *
9
+ * Two properties make it different from `Agent`:
10
+ *
11
+ * - **Stateless.** Nothing survives the call. A craft's state is queryable
12
+ * from the workbook, so the model re-reads the live position every time
13
+ * rather than replaying a history that can go stale.
14
+ * - **Read-only.** The loop is offered the craft's own `@mutates none` tools
15
+ * and nothing else — no `EDIT_TOOLS`, no workbook surface. The model
16
+ * gathers; the craft decides what to do with the answer. That is what keeps
17
+ * an `askAi` retryable and cancellable: abandoning one mid-loop cannot have
18
+ * half-changed the workbook.
19
+ */
20
+ import type { LlmClient } from './agent/loop.js';
21
+ import { type JSONSchema, type Tool, type ToolContext } from './tool.js';
22
+ /** One question a craft can ask, as `craftsmith` extracted it from `@aiRole`. */
23
+ export interface AiRole {
24
+ /** Used only in error messages here; the host selects roles by it. */
25
+ name: string;
26
+ /** Sent verbatim as the (cached) system prompt. */
27
+ system: string;
28
+ /**
29
+ * Becomes the `reply` tool's input schema. Only checked shallowly: an
30
+ * object, `required` keys present, top-level primitive `type`s and
31
+ * `enum`s — nested shapes are the craft's to validate.
32
+ */
33
+ replySchema: JSONSchema;
34
+ }
35
+ export interface AskAiParams {
36
+ llm: LlmClient;
37
+ model: string;
38
+ role: AiRole;
39
+ /** The question. Not the state — that is what the tools are for. */
40
+ input: string;
41
+ /** The craft's read-only tools. A mutating tool here is a caller bug. */
42
+ tools: readonly Tool[];
43
+ /** Context handed to a dispatched craft tool; its `signal` also aborts us. */
44
+ ctx: ToolContext;
45
+ /** Per-response output cap. Default 16384 — see DEFAULT_MAX_TOKENS. */
46
+ max_tokens?: number;
47
+ /** LLM round-trips allowed, the answering one included. Default 8. */
48
+ max_iterations?: number;
49
+ }
50
+ export declare class AskAiError extends Error {
51
+ /** The turn the loop gave up on, for a host that wants to log it. */
52
+ readonly detail?: unknown | undefined;
53
+ constructor(message: string,
54
+ /** The turn the loop gave up on, for a host that wants to log it. */
55
+ detail?: unknown | undefined);
56
+ }
57
+ /**
58
+ * Run one question to completion and return the model's parsed answer.
59
+ *
60
+ * Throws {@link AskAiError} if `tools` contains a mutating tool (before any
61
+ * request), a response stops at max_tokens, the model answers in prose twice,
62
+ * answers in the wrong shape twice, or runs past `max_iterations`. Anything
63
+ * else is not wrapped: an abort of `ctx.signal` throws the signal's reason
64
+ * and an `LlmClient` rejection propagates. A craft tool that throws does not
65
+ * end the call; the model gets the message as an `is_error` result.
66
+ */
67
+ export declare function askAi<T = unknown>(params: AskAiParams): Promise<T>;
@@ -0,0 +1,214 @@
1
+ /**
2
+ * `askAi` — the craft→AI direction.
3
+ *
4
+ * The three craft faces all run inward: something else calls the craft. This is
5
+ * the one that runs outward, so a craft can put a question to a model and get a
6
+ * schema-valid answer: a chess board asking for the opponent's move, a
7
+ * simulator asking what to look at next.
8
+ *
9
+ * Two properties make it different from `Agent`:
10
+ *
11
+ * - **Stateless.** Nothing survives the call. A craft's state is queryable
12
+ * from the workbook, so the model re-reads the live position every time
13
+ * rather than replaying a history that can go stale.
14
+ * - **Read-only.** The loop is offered the craft's own `@mutates none` tools
15
+ * and nothing else — no `EDIT_TOOLS`, no workbook surface. The model
16
+ * gathers; the craft decides what to do with the answer. That is what keeps
17
+ * an `askAi` retryable and cancellable: abandoning one mid-loop cannot have
18
+ * half-changed the workbook.
19
+ */
20
+ import { toLlmTool, } from './tool.js';
21
+ /** The reply tool's id. Not namespaced — it is synthetic, not a craft tool. */
22
+ const REPLY = 'reply';
23
+ /** Safety and cost ceiling: how many times the model may go back for state. */
24
+ const DEFAULT_MAX_ITERATIONS = 8;
25
+ /**
26
+ * Room for one turn's output.
27
+ *
28
+ * Generous because a reasoning model spends nearly all of it on a `thinking`
29
+ * block before it emits anything. Measured against claude-opus-5 deciding a
30
+ * single 四象 draft pick: 1024 was cut off mid-thought every time, and 4096
31
+ * still was. The role's own prompt is what drives this — it says to work the
32
+ * position out rather than be handed a score — so the budget has to cover
33
+ * reasoning, not the answer, which is a few dozen tokens.
34
+ *
35
+ * A role that wants a tighter leash passes its own `max_tokens`.
36
+ */
37
+ const DEFAULT_MAX_TOKENS = 16384;
38
+ export class AskAiError extends Error {
39
+ constructor(message,
40
+ /** The turn the loop gave up on, for a host that wants to log it. */
41
+ detail) {
42
+ super(message);
43
+ this.detail = detail;
44
+ this.name = 'AskAiError';
45
+ }
46
+ }
47
+ /**
48
+ * The reply schema as one more tool in the set. It cannot be forced with
49
+ * `tool_choice` — the model has to be free to read first — so termination is
50
+ * this loop's job: it ends when the model calls `reply`.
51
+ */
52
+ function replyTool(role) {
53
+ return {
54
+ name: REPLY,
55
+ description: 'Give your final answer. Call this exactly once, when you have ' +
56
+ 'read everything you need.',
57
+ input_schema: { type: 'object', ...role.replySchema },
58
+ };
59
+ }
60
+ /**
61
+ * Shallow structural check against the reply schema — enough to catch a model
62
+ * that answered with the wrong shape, which is the failure this guards. Deep
63
+ * validation is the craft's business; it knows what a legal move is.
64
+ */
65
+ function schemaViolations(schema, value) {
66
+ var _a, _b;
67
+ if (typeof value !== 'object' || value === null || Array.isArray(value))
68
+ return ['the answer must be an object'];
69
+ const obj = value;
70
+ const out = [];
71
+ for (const key of (_a = schema.required) !== null && _a !== void 0 ? _a : [])
72
+ if (obj[key] === undefined)
73
+ out.push(`"${key}" is required but missing`);
74
+ for (const [key, spec] of Object.entries((_b = schema.properties) !== null && _b !== void 0 ? _b : {})) {
75
+ const v = obj[key];
76
+ if (v === undefined)
77
+ continue;
78
+ const want = spec.type;
79
+ const got = Array.isArray(v) ? 'array' : typeof v;
80
+ if (want === 'number' && got !== 'number')
81
+ out.push(`"${key}" must be a number, got ${got}`);
82
+ else if (want === 'string' && got !== 'string')
83
+ out.push(`"${key}" must be a string, got ${got}`);
84
+ else if (want === 'boolean' && got !== 'boolean')
85
+ out.push(`"${key}" must be a boolean, got ${got}`);
86
+ else if (want === 'array' && got !== 'array')
87
+ out.push(`"${key}" must be an array, got ${got}`);
88
+ const allowed = spec.enum;
89
+ if (allowed && !allowed.includes(v))
90
+ out.push(`"${key}" must be one of ${allowed.join(', ')}`);
91
+ }
92
+ return out;
93
+ }
94
+ function textOf(content) {
95
+ return content
96
+ .filter((b) => b.type === 'text')
97
+ .map((b) => b.text)
98
+ .join(' ')
99
+ .trim();
100
+ }
101
+ /**
102
+ * Run one question to completion and return the model's parsed answer.
103
+ *
104
+ * Throws {@link AskAiError} if `tools` contains a mutating tool (before any
105
+ * request), a response stops at max_tokens, the model answers in prose twice,
106
+ * answers in the wrong shape twice, or runs past `max_iterations`. Anything
107
+ * else is not wrapped: an abort of `ctx.signal` throws the signal's reason
108
+ * and an `LlmClient` rejection propagates. A craft tool that throws does not
109
+ * end the call; the model gets the message as an `is_error` result.
110
+ */
111
+ export async function askAi(params) {
112
+ var _a, _b, _c, _d;
113
+ const { llm, model, role, input, tools, ctx } = params;
114
+ const maxIterations = (_a = params.max_iterations) !== null && _a !== void 0 ? _a : DEFAULT_MAX_ITERATIONS;
115
+ const mutating = tools.filter((t) => t.mutates);
116
+ if (mutating.length)
117
+ throw new AskAiError(`role "${role.name}" was given mutating tool(s) ` +
118
+ `${mutating
119
+ .map((t) => t.name)
120
+ .join(', ')}; an askAi loop reads only`);
121
+ const byId = new Map(tools.map((t) => [`${t.namespace}__${t.name}`, t]));
122
+ const llmTools = [...tools.map(toLlmTool), replyTool(role)];
123
+ const system = [
124
+ { type: 'text', text: role.system, cache_control: { type: 'ephemeral' } },
125
+ ];
126
+ const messages = [{ role: 'user', content: input }];
127
+ let retriedShape = false;
128
+ let noAnswerNudges = 0;
129
+ for (let i = 0; i < maxIterations; i++) {
130
+ ctx.signal.throwIfAborted();
131
+ const res = await llm.createMessage({
132
+ model,
133
+ system,
134
+ tools: llmTools,
135
+ messages,
136
+ max_tokens: (_b = params.max_tokens) !== null && _b !== void 0 ? _b : DEFAULT_MAX_TOKENS,
137
+ signal: ctx.signal,
138
+ });
139
+ messages.push({ role: 'assistant', content: res.content });
140
+ // A truncated turn can carry no usable tool call, so left alone it
141
+ // silently spends an iteration and the loop ends up reporting that the
142
+ // model never answered — which is not what went wrong.
143
+ if (res.stop_reason === 'max_tokens')
144
+ throw new AskAiError(`role "${role.name}" was cut off at max_tokens ` +
145
+ `(${(_c = params.max_tokens) !== null && _c !== void 0 ? _c : DEFAULT_MAX_TOKENS}) before it ` +
146
+ `could answer; raise max_tokens for this role`, textOf(res.content));
147
+ const calls = res.content.filter((b) => b.type === 'tool_use');
148
+ const answer = calls.find((c) => c.name === REPLY);
149
+ if (answer) {
150
+ const bad = schemaViolations(role.replySchema, answer.input);
151
+ if (!bad.length)
152
+ return answer.input;
153
+ if (retriedShape)
154
+ throw new AskAiError(`role "${role.name}" answered in the wrong shape: ${bad.join('; ')}`, answer.input);
155
+ retriedShape = true;
156
+ messages.push({
157
+ role: 'user',
158
+ content: [
159
+ {
160
+ type: 'tool_result',
161
+ tool_use_id: answer.id,
162
+ content: `That answer does not fit: ${bad.join('; ')}. Call ${REPLY} again, corrected.`,
163
+ is_error: true,
164
+ },
165
+ ],
166
+ });
167
+ continue;
168
+ }
169
+ if (!calls.length) {
170
+ // A turn of prose instead of an answer. Nudge once; a model that
171
+ // still will not use the tool is not going to.
172
+ if (noAnswerNudges++ > 0)
173
+ throw new AskAiError(`role "${role.name}" never called ${REPLY}`, textOf(res.content));
174
+ messages.push({
175
+ role: 'user',
176
+ content: `Answer by calling the ${REPLY} tool.`,
177
+ });
178
+ continue;
179
+ }
180
+ const results = [];
181
+ for (const call of calls) {
182
+ const tool = byId.get(call.name);
183
+ if (!tool) {
184
+ results.push({
185
+ type: 'tool_result',
186
+ tool_use_id: call.id,
187
+ content: `No tool named ${call.name}.`,
188
+ is_error: true,
189
+ });
190
+ continue;
191
+ }
192
+ try {
193
+ const out = await tool.handler(call.input, ctx);
194
+ results.push({
195
+ type: 'tool_result',
196
+ tool_use_id: call.id,
197
+ content: JSON.stringify((_d = out.data) !== null && _d !== void 0 ? _d : null),
198
+ });
199
+ }
200
+ catch (e) {
201
+ // The model gets the reason and can try a different read; only
202
+ // the loop's own limits end the call.
203
+ results.push({
204
+ type: 'tool_result',
205
+ tool_use_id: call.id,
206
+ content: e instanceof Error ? e.message : String(e),
207
+ is_error: true,
208
+ });
209
+ }
210
+ }
211
+ messages.push({ role: 'user', content: results });
212
+ }
213
+ throw new AskAiError(`role "${role.name}" did not answer within ${maxIterations} steps`);
214
+ }
@@ -0,0 +1 @@
1
+ export {};
@@ -0,0 +1,201 @@
1
+ import { describe, it, expect, vi } from 'vitest';
2
+ import { askAi, AskAiError } from './craft-ai.js';
3
+ const ROLE = {
4
+ name: 'opponent',
5
+ system: 'You play chess as Black.',
6
+ replySchema: {
7
+ type: 'object',
8
+ properties: { uci: { type: 'string' }, plies: { type: 'number' } },
9
+ required: ['uci'],
10
+ },
11
+ };
12
+ /**
13
+ * Replays a canned script of model turns, snapshotting each request. The loop
14
+ * appends to one `messages` array as it goes, so recording the reference would
15
+ * make every turn look like the last one.
16
+ */
17
+ function fakeLlm(turns) {
18
+ const sent = [];
19
+ let i = 0;
20
+ return {
21
+ sent,
22
+ async createMessage(params) {
23
+ sent.push({
24
+ system: structuredClone(params.system),
25
+ tools: params.tools.map((t) => ({ name: t.name })),
26
+ messages: structuredClone(params.messages),
27
+ });
28
+ const content = turns[i++];
29
+ if (!content)
30
+ throw new Error('fake llm ran out of turns');
31
+ return {
32
+ content,
33
+ stop_reason: content.some((b) => b.type === 'tool_use')
34
+ ? 'tool_use'
35
+ : 'end_turn',
36
+ };
37
+ },
38
+ };
39
+ }
40
+ const use = (name, input = {}, id = name) => ({
41
+ type: 'tool_use',
42
+ id,
43
+ name,
44
+ input,
45
+ });
46
+ const say = (text) => ({ type: 'text', text });
47
+ function readTool(name, data, mutates = false) {
48
+ return {
49
+ namespace: 'chess',
50
+ name,
51
+ description: name,
52
+ inputSchema: { type: 'object', properties: {} },
53
+ mutates,
54
+ handler: vi.fn(async () => ({ data })),
55
+ };
56
+ }
57
+ function ctx(signal = new AbortController().signal) {
58
+ return {
59
+ workbook: {},
60
+ signal,
61
+ confirm: async () => true,
62
+ log: () => { },
63
+ };
64
+ }
65
+ const run = (llm, tools = [], over) => askAi({
66
+ llm,
67
+ model: 'test',
68
+ role: ROLE,
69
+ input: 'Your move.',
70
+ tools,
71
+ ctx: ctx(),
72
+ ...over,
73
+ });
74
+ describe('askAi', () => {
75
+ it('returns the reply the model gives', async () => {
76
+ const llm = fakeLlm([[use('reply', { uci: 'g8f6' })]]);
77
+ await expect(run(llm)).resolves.toEqual({ uci: 'g8f6' });
78
+ });
79
+ it('lets the model read through the craft tools before answering', async () => {
80
+ const position = readTool('get_position', { fen: 'startpos' });
81
+ const legal = readTool('get_legal_moves', { moves: ['g8f6'] });
82
+ const llm = fakeLlm([
83
+ [use('chess__get_position')],
84
+ [use('chess__get_legal_moves')],
85
+ [use('reply', { uci: 'g8f6' })],
86
+ ]);
87
+ await expect(run(llm, [position, legal])).resolves.toEqual({
88
+ uci: 'g8f6',
89
+ });
90
+ expect(position.handler).toHaveBeenCalledOnce();
91
+ expect(legal.handler).toHaveBeenCalledOnce();
92
+ // The tool result has to come back as the model's next input.
93
+ expect(JSON.stringify(llm.sent[1].messages)).toContain('startpos');
94
+ });
95
+ it('offers the craft tools plus a reply tool, and nothing else', async () => {
96
+ const llm = fakeLlm([[use('reply', { uci: 'g8f6' })]]);
97
+ await run(llm, [readTool('get_position', {})]);
98
+ expect(llm.sent[0].tools.map((t) => t.name)).toEqual([
99
+ 'chess__get_position',
100
+ 'reply',
101
+ ]);
102
+ });
103
+ it('refuses a mutating tool rather than letting the model write', async () => {
104
+ const llm = fakeLlm([[use('reply', { uci: 'g8f6' })]]);
105
+ await expect(run(llm, [readTool('apply_move', {}, true)])).rejects.toThrow(/reads only/);
106
+ });
107
+ it('re-asks once when the reply has the wrong shape', async () => {
108
+ const llm = fakeLlm([
109
+ [use('reply', { plies: 3 })], // missing the required uci
110
+ [use('reply', { uci: 'g8f6' })],
111
+ ]);
112
+ await expect(run(llm)).resolves.toEqual({ uci: 'g8f6' });
113
+ expect(JSON.stringify(llm.sent[1].messages)).toContain('is required but missing');
114
+ });
115
+ it('gives up after a second wrong shape, saying what was wrong', async () => {
116
+ const llm = fakeLlm([
117
+ [use('reply', { uci: 1 })],
118
+ [use('reply', { uci: 2 })],
119
+ ]);
120
+ await expect(run(llm)).rejects.toThrow(/"uci" must be a string, got number/);
121
+ });
122
+ it('nudges a model that answers in prose, then gives up', async () => {
123
+ const llm = fakeLlm([[say("I'd play Nf6")], [say('Nf6, definitely')]]);
124
+ await expect(run(llm)).rejects.toThrow(/never called reply/);
125
+ expect(JSON.stringify(llm.sent[1].messages)).toContain('Answer by calling the reply tool');
126
+ });
127
+ it('hands a failing tool back to the model instead of ending the call', async () => {
128
+ const boom = {
129
+ ...readTool('get_position', null),
130
+ handler: async () => {
131
+ throw new Error('board not initialized');
132
+ },
133
+ };
134
+ const llm = fakeLlm([
135
+ [use('chess__get_position')],
136
+ [use('reply', { uci: 'g8f6' })],
137
+ ]);
138
+ await expect(run(llm, [boom])).resolves.toEqual({ uci: 'g8f6' });
139
+ expect(JSON.stringify(llm.sent[1].messages)).toContain('board not initialized');
140
+ });
141
+ it('tells the model when it invents a tool', async () => {
142
+ const llm = fakeLlm([
143
+ [use('chess__get_evaluation')],
144
+ [use('reply', { uci: 'g8f6' })],
145
+ ]);
146
+ await expect(run(llm, [readTool('get_position', {})])).resolves.toEqual({
147
+ uci: 'g8f6',
148
+ });
149
+ expect(JSON.stringify(llm.sent[1].messages)).toContain('No tool named chess__get_evaluation');
150
+ });
151
+ it('stops at max_iterations rather than looping on reads', async () => {
152
+ const llm = fakeLlm(Array.from({ length: 10 }, () => [use('chess__get_position')]));
153
+ await expect(run(llm, [readTool('get_position', {})], { max_iterations: 3 })).rejects.toThrow(/did not answer within 3 steps/);
154
+ expect(llm.sent).toHaveLength(3);
155
+ });
156
+ it('aborts when the craft cancels', async () => {
157
+ const ac = new AbortController();
158
+ ac.abort();
159
+ const llm = fakeLlm([[use('reply', { uci: 'g8f6' })]]);
160
+ await expect(askAi({
161
+ llm,
162
+ model: 'test',
163
+ role: ROLE,
164
+ input: 'Your move.',
165
+ tools: [],
166
+ ctx: ctx(ac.signal),
167
+ })).rejects.toThrow();
168
+ expect(llm.sent).toHaveLength(0);
169
+ });
170
+ it('sends the role system prompt as a cacheable block', async () => {
171
+ const llm = fakeLlm([[use('reply', { uci: 'g8f6' })]]);
172
+ await run(llm);
173
+ expect(llm.sent[0].system).toEqual([
174
+ {
175
+ type: 'text',
176
+ text: 'You play chess as Black.',
177
+ cache_control: { type: 'ephemeral' },
178
+ },
179
+ ]);
180
+ });
181
+ it('carries no state between calls', async () => {
182
+ const first = fakeLlm([[use('reply', { uci: 'g8f6' })]]);
183
+ const second = fakeLlm([[use('reply', { uci: 'b8c6' })]]);
184
+ await run(first);
185
+ await run(second);
186
+ // The second call starts from the question alone.
187
+ expect(second.sent[0].messages).toEqual([
188
+ { role: 'user', content: 'Your move.' },
189
+ ]);
190
+ });
191
+ });
192
+ describe('AskAiError', () => {
193
+ it('carries the turn it gave up on', async () => {
194
+ const llm = fakeLlm([[say('nope')], [say('still nope')]]);
195
+ await run(llm).catch((e) => {
196
+ expect(e).toBeInstanceOf(AskAiError);
197
+ expect(e.detail).toBe('still nope');
198
+ });
199
+ expect.assertions(2);
200
+ });
201
+ });