crewly 1.20.35 → 1.20.40
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/config/skills/_common/lib.sh +6 -0
- package/config/skills/agent/core/calendar-create/SKILL.md +10 -0
- package/config/skills/agent/core/calendar-create/execute.sh +6 -0
- package/config/skills/agent/core/calendar-list/SKILL.md +10 -0
- package/config/skills/agent/core/calendar-list/execute.sh +6 -0
- package/config/skills/agent/core/docs-read/SKILL.md +10 -0
- package/config/skills/agent/core/docs-read/execute.sh +6 -1
- package/config/skills/agent/core/docs-write/SKILL.md +10 -0
- package/config/skills/agent/core/docs-write/execute.sh +6 -1
- package/config/skills/agent/core/drive-read/SKILL.md +10 -0
- package/config/skills/agent/core/drive-read/execute.sh +6 -1
- package/config/skills/agent/core/drive-search/SKILL.md +10 -0
- package/config/skills/agent/core/drive-search/execute.sh +6 -1
- package/config/skills/agent/core/drive-upload/SKILL.md +10 -0
- package/config/skills/agent/core/drive-upload/execute.sh +6 -1
- package/config/skills/agent/core/gmail-read/SKILL.md +10 -0
- package/config/skills/agent/core/gmail-read/execute.sh +6 -0
- package/config/skills/agent/core/gmail-search/SKILL.md +10 -0
- package/config/skills/agent/core/gmail-search/execute.sh +6 -0
- package/config/skills/agent/core/gmail-send/SKILL.md +10 -0
- package/config/skills/agent/core/gmail-send/execute.sh +6 -0
- package/config/skills/agent/core/sheets-read/SKILL.md +10 -0
- package/config/skills/agent/core/sheets-read/execute.sh +6 -1
- package/config/skills/agent/core/sheets-write/SKILL.md +10 -0
- package/config/skills/agent/core/sheets-write/execute.sh +6 -1
- package/config/skills/agent/core/slides-create/SKILL.md +10 -0
- package/config/skills/agent/core/slides-create/execute.sh +6 -1
- package/config/skills/agent/core/slides-read/SKILL.md +10 -0
- package/config/skills/agent/core/slides-read/execute.sh +6 -1
- package/config/slack-app-manifest.json +16 -9
- package/dist/backend/backend/src/constants.d.ts +18 -4
- package/dist/backend/backend/src/constants.d.ts.map +1 -1
- package/dist/backend/backend/src/constants.js +16 -4
- package/dist/backend/backend/src/constants.js.map +1 -1
- package/dist/backend/backend/src/controllers/google/google.controller.d.ts +8 -0
- package/dist/backend/backend/src/controllers/google/google.controller.d.ts.map +1 -1
- package/dist/backend/backend/src/controllers/google/google.controller.js +137 -37
- package/dist/backend/backend/src/controllers/google/google.controller.js.map +1 -1
- package/dist/backend/backend/src/controllers/google/google.routes.d.ts +2 -1
- package/dist/backend/backend/src/controllers/google/google.routes.d.ts.map +1 -1
- package/dist/backend/backend/src/controllers/google/google.routes.js +4 -2
- package/dist/backend/backend/src/controllers/google/google.routes.js.map +1 -1
- package/dist/backend/backend/src/controllers/slack/slack-error.utils.d.ts +46 -0
- package/dist/backend/backend/src/controllers/slack/slack-error.utils.d.ts.map +1 -0
- package/dist/backend/backend/src/controllers/slack/slack-error.utils.js +54 -0
- package/dist/backend/backend/src/controllers/slack/slack-error.utils.js.map +1 -0
- package/dist/backend/backend/src/controllers/slack/slack.controller.d.ts.map +1 -1
- package/dist/backend/backend/src/controllers/slack/slack.controller.js +5 -12
- package/dist/backend/backend/src/controllers/slack/slack.controller.js.map +1 -1
- package/dist/backend/backend/src/services/google/google-api.client.d.ts +23 -2
- package/dist/backend/backend/src/services/google/google-api.client.d.ts.map +1 -1
- package/dist/backend/backend/src/services/google/google-api.client.js +5 -2
- package/dist/backend/backend/src/services/google/google-api.client.js.map +1 -1
- package/dist/backend/backend/src/services/google/google-workspace-token.service.d.ts +61 -11
- package/dist/backend/backend/src/services/google/google-workspace-token.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/google/google-workspace-token.service.js +108 -31
- package/dist/backend/backend/src/services/google/google-workspace-token.service.js.map +1 -1
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.js +27 -4
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
- package/dist/backend/build-info.json +2 -2
- package/dist/cli/backend/src/constants.d.ts +18 -4
- package/dist/cli/backend/src/constants.d.ts.map +1 -1
- package/dist/cli/backend/src/constants.js +16 -4
- package/dist/cli/backend/src/constants.js.map +1 -1
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.js +27 -4
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
- package/frontend/dist/assets/{index-e079a375.js → index-e7785269.js} +267 -267
- package/frontend/dist/index.html +1 -1
- package/package.json +1 -1
- package/packages/crewly-agent/src/runtime/agent-runner.service.test.ts +10 -1
- package/packages/crewly-agent/src/runtime/agent-runner.service.ts +179 -3
- package/packages/crewly-agent/src/runtime/text-tool-calls.test.ts +144 -0
- package/packages/crewly-agent/src/runtime/text-tool-calls.ts +316 -0
- package/packages/crewly-agent/src/runtime/text-tool-salvage.test.ts +190 -0
- package/packages/crewly-agent/src/runtime/types.ts +6 -0
|
@@ -0,0 +1,316 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Recover tool calls a model wrote as text.
|
|
3
|
+
*
|
|
4
|
+
* A weak model sometimes *writes* its function-call envelope into the text
|
|
5
|
+
* channel instead of emitting a real tool call. The provider sees ordinary
|
|
6
|
+
* prose, the AI SDK sees no tool call, the step ends — so nothing runs, and
|
|
7
|
+
* the user is shown a wall of markup where an answer should be. deepseek-chat
|
|
8
|
+
* does this with its own separators:
|
|
9
|
+
*
|
|
10
|
+
* `<||DSML|| invoke name="Bash">` … `<||DSML|| parameter name="command" …>`
|
|
11
|
+
*
|
|
12
|
+
* which is Claude's `invoke`/`parameter` grammar with `||DSML|| ` (U+FF5C
|
|
13
|
+
* pipes) where the namespace prefix goes. Nothing in Crewly teaches that
|
|
14
|
+
* format — the model emits it on its own (2026-09-19, #crewly-support).
|
|
15
|
+
*
|
|
16
|
+
* Stripping the markup hides the symptom and still loses the work. This
|
|
17
|
+
* module instead *parses* the envelope so the runtime can execute what the
|
|
18
|
+
* model meant to call and hand the results back, turning a dead turn into a
|
|
19
|
+
* working one. Matching is by **shape**, never by a list of known names: the
|
|
20
|
+
* prefix is whatever the model felt like emitting.
|
|
21
|
+
*
|
|
22
|
+
* @module runtime/text-tool-calls
|
|
23
|
+
*/
|
|
24
|
+
|
|
25
|
+
/** One tool call recovered from text. Values are raw, as written. */
|
|
26
|
+
export interface TextToolCall {
|
|
27
|
+
/** Tool name from the `name="…"` attribute. */
|
|
28
|
+
toolName: string;
|
|
29
|
+
/** Parameter name → raw text value. */
|
|
30
|
+
args: Record<string, string>;
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
/** Result of {@link parseTextToolCalls}. */
|
|
34
|
+
export interface ParsedTextToolCalls {
|
|
35
|
+
/** The calls found, in the order they were written. */
|
|
36
|
+
calls: TextToolCall[];
|
|
37
|
+
/** The same text with every envelope removed and whitespace tidied. */
|
|
38
|
+
text: string;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
/** Anything with a Zod-style `safeParse` — kept duck-typed so this module stays dependency-free. */
|
|
42
|
+
export interface SchemaLike {
|
|
43
|
+
safeParse(value: unknown): { success: boolean; data?: unknown; error?: { message?: string } };
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
/** Result of {@link coerceArgs}. */
|
|
47
|
+
export interface CoercedArgs {
|
|
48
|
+
/** Arguments to pass to the tool. */
|
|
49
|
+
args: Record<string, unknown>;
|
|
50
|
+
/** Why the schema rejected them, when it did. */
|
|
51
|
+
error?: string;
|
|
52
|
+
}
|
|
53
|
+
|
|
54
|
+
/**
|
|
55
|
+
* A tag like `<invoke …>`, `</invoke>`, `<||DSML|| parameter …>` — any
|
|
56
|
+
* junk or namespace before the name, anything but `<`/`>` after it.
|
|
57
|
+
*
|
|
58
|
+
* @param name - Tag-name pattern
|
|
59
|
+
* @param closing - Match the closing form
|
|
60
|
+
* @returns The pattern source
|
|
61
|
+
*/
|
|
62
|
+
function tagSource(name: string, closing: boolean): string {
|
|
63
|
+
return `<${closing ? '\\s*/' : ''}[^<>]{0,40}?${name}\\b[^<>]*>`;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
/** Wrapper the model puts around a batch of calls: `function_calls`, `tool_calls`, or a bare `calls`. */
|
|
67
|
+
const WRAPPER_NAME = '(?:function_calls|tool_calls|(?<![A-Za-z_])calls)';
|
|
68
|
+
|
|
69
|
+
const INVOKE_OPEN = new RegExp(tagSource('invoke', false), 'gi');
|
|
70
|
+
const INVOKE_CLOSE = new RegExp(tagSource('invoke', true), 'i');
|
|
71
|
+
const PARAM_OPEN = new RegExp(tagSource('parameter', false), 'gi');
|
|
72
|
+
const PARAM_CLOSE = new RegExp(tagSource('parameter', true), 'i');
|
|
73
|
+
|
|
74
|
+
/** `name="x"` / `name='x'` / `name=x` inside a tag. */
|
|
75
|
+
const NAME_ATTR = /\bname\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'<>]+))/i;
|
|
76
|
+
|
|
77
|
+
/** Well-formed `<x>…</x>` blocks, for removal. */
|
|
78
|
+
const BLOCK_PATTERNS = [WRAPPER_NAME, 'invoke'].map(
|
|
79
|
+
(name) => new RegExp(`${tagSource(name, false)}[\\s\\S]*?${tagSource(name, true)}`, 'gi'),
|
|
80
|
+
);
|
|
81
|
+
|
|
82
|
+
/** An opening tag with no close — the output was cut off mid-call. */
|
|
83
|
+
const UNCLOSED_PATTERN = new RegExp(`(?:${tagSource(WRAPPER_NAME, false)}|${tagSource('invoke', false)})[\\s\\S]*$`, 'i');
|
|
84
|
+
|
|
85
|
+
/** Any leftover lone tag. */
|
|
86
|
+
const LONE_TAG_PATTERN = new RegExp(
|
|
87
|
+
[WRAPPER_NAME, 'invoke', 'parameter'].map((name) => `${tagSource(name, false)}|${tagSource(name, true)}`).join('|'),
|
|
88
|
+
'gi',
|
|
89
|
+
);
|
|
90
|
+
|
|
91
|
+
/** A value that is plainly JSON rather than prose. */
|
|
92
|
+
const JSONISH = /^(?:true|false|null|-?\d+(?:\.\d+)?(?:[eE][+-]?\d+)?|\{[\s\S]*\}|\[[\s\S]*\])$/;
|
|
93
|
+
|
|
94
|
+
/**
|
|
95
|
+
* Read the `name="…"` attribute out of an opening tag.
|
|
96
|
+
*
|
|
97
|
+
* @param tag - The whole tag, angle brackets included
|
|
98
|
+
* @returns The name, or '' when the tag has none
|
|
99
|
+
*/
|
|
100
|
+
function nameOf(tag: string): string {
|
|
101
|
+
const m = NAME_ATTR.exec(tag);
|
|
102
|
+
return (m?.[1] ?? m?.[2] ?? m?.[3] ?? '').trim();
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
/**
|
|
106
|
+
* Split text into segments, marking the fenced ones.
|
|
107
|
+
*
|
|
108
|
+
* Markup inside a ``` fence is an example the agent meant to show — it is
|
|
109
|
+
* neither executed nor stripped.
|
|
110
|
+
*
|
|
111
|
+
* @param input - Raw text
|
|
112
|
+
* @returns Segments; odd indexes are fenced code
|
|
113
|
+
*/
|
|
114
|
+
function splitOnFences(input: string): string[] {
|
|
115
|
+
return input.split(/(```[\s\S]*?```)/g);
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/**
|
|
119
|
+
* Parse the `parameter` blocks inside one `invoke` body.
|
|
120
|
+
*
|
|
121
|
+
* A block runs to its closing tag, or — when the model never wrote one — to
|
|
122
|
+
* the next parameter or the end of the body, so a truncated call still yields
|
|
123
|
+
* the arguments it managed to write.
|
|
124
|
+
*
|
|
125
|
+
* @param body - Text between the invoke tags
|
|
126
|
+
* @returns Parameter name → raw value
|
|
127
|
+
*/
|
|
128
|
+
function parseParams(body: string): Record<string, string> {
|
|
129
|
+
const args: Record<string, string> = {};
|
|
130
|
+
PARAM_OPEN.lastIndex = 0;
|
|
131
|
+
const opens: Array<{ name: string; from: number }> = [];
|
|
132
|
+
for (let m = PARAM_OPEN.exec(body); m; m = PARAM_OPEN.exec(body)) {
|
|
133
|
+
opens.push({ name: nameOf(m[0]), from: m.index + m[0].length });
|
|
134
|
+
}
|
|
135
|
+
opens.forEach((open, i) => {
|
|
136
|
+
const until = i + 1 < opens.length ? body.lastIndexOf('<', opens[i + 1].from) : body.length;
|
|
137
|
+
const slice = body.slice(open.from, Math.max(until, open.from));
|
|
138
|
+
const close = PARAM_CLOSE.exec(slice);
|
|
139
|
+
const value = close ? slice.slice(0, close.index) : slice;
|
|
140
|
+
if (open.name) args[open.name] = value.replace(/^\r?\n/, '').replace(/\s+$/, '');
|
|
141
|
+
});
|
|
142
|
+
return args;
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
/**
|
|
146
|
+
* Find every tool call written as text, and return the text without them.
|
|
147
|
+
*
|
|
148
|
+
* @param raw - The model's text output
|
|
149
|
+
* @returns The recovered calls and the cleaned text
|
|
150
|
+
*
|
|
151
|
+
* @example
|
|
152
|
+
* parseTextToolCalls('<invoke name="bash_exec"><parameter name="command">ls</parameter></invoke>')
|
|
153
|
+
* // → { calls: [{ toolName: 'bash_exec', args: { command: 'ls' } }], text: '' }
|
|
154
|
+
*/
|
|
155
|
+
export function parseTextToolCalls(raw: string): ParsedTextToolCalls {
|
|
156
|
+
const input = raw ?? '';
|
|
157
|
+
if (!input) return { calls: [], text: '' };
|
|
158
|
+
|
|
159
|
+
const calls: TextToolCall[] = [];
|
|
160
|
+
const cleaned = splitOnFences(input)
|
|
161
|
+
.map((segment, i) => {
|
|
162
|
+
if (i % 2 === 1) return segment; // fenced code — an example, not a call
|
|
163
|
+
|
|
164
|
+
INVOKE_OPEN.lastIndex = 0;
|
|
165
|
+
const opens: Array<{ name: string; from: number }> = [];
|
|
166
|
+
for (let m = INVOKE_OPEN.exec(segment); m; m = INVOKE_OPEN.exec(segment)) {
|
|
167
|
+
opens.push({ name: nameOf(m[0]), from: m.index + m[0].length });
|
|
168
|
+
}
|
|
169
|
+
opens.forEach((open, idx) => {
|
|
170
|
+
// An unclosed invoke runs to the next one, or to the end of the text.
|
|
171
|
+
const until = idx + 1 < opens.length ? segment.lastIndexOf('<', opens[idx + 1].from) : segment.length;
|
|
172
|
+
const slice = segment.slice(open.from, Math.max(until, open.from));
|
|
173
|
+
const close = INVOKE_CLOSE.exec(slice);
|
|
174
|
+
const body = close ? slice.slice(0, close.index) : slice;
|
|
175
|
+
if (open.name) calls.push({ toolName: open.name, args: parseParams(body) });
|
|
176
|
+
});
|
|
177
|
+
|
|
178
|
+
let out = segment;
|
|
179
|
+
for (const pattern of BLOCK_PATTERNS) out = out.replace(pattern, '');
|
|
180
|
+
out = out.replace(UNCLOSED_PATTERN, '');
|
|
181
|
+
return out.replace(LONE_TAG_PATTERN, '');
|
|
182
|
+
})
|
|
183
|
+
.join('');
|
|
184
|
+
|
|
185
|
+
const text = cleaned
|
|
186
|
+
.split('\n')
|
|
187
|
+
.map((line) => line.replace(/[ \t]+$/, ''))
|
|
188
|
+
.join('\n')
|
|
189
|
+
.replace(/\n{3,}/g, '\n\n')
|
|
190
|
+
.trim();
|
|
191
|
+
|
|
192
|
+
return { calls, text };
|
|
193
|
+
}
|
|
194
|
+
|
|
195
|
+
/**
|
|
196
|
+
* Whether text contains something that looks like a tool-call envelope.
|
|
197
|
+
*
|
|
198
|
+
* @param raw - Candidate text
|
|
199
|
+
* @returns True when an `invoke` or wrapper tag is present
|
|
200
|
+
*/
|
|
201
|
+
export function hasTextToolCalls(raw: string): boolean {
|
|
202
|
+
return new RegExp(`${tagSource(WRAPPER_NAME, false)}|${tagSource('invoke', false)}`, 'i').test(raw ?? '');
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
/**
|
|
206
|
+
* Names other harnesses use for tools Crewly spells differently.
|
|
207
|
+
*
|
|
208
|
+
* A model that has seen Claude Code reaches for `Bash` and `Read`; ours are
|
|
209
|
+
* `bash_exec` and `read_file`. Observed live on the first turn after the
|
|
210
|
+
* salvage shipped: deepseek-chat wrote `Bash` twice, was told the name did
|
|
211
|
+
* not exist, and only then used real names (2026-09-19). Resolving the alias
|
|
212
|
+
* turns that wasted round into a working one. This maps onto tools the agent
|
|
213
|
+
* already has — it never grants one it does not.
|
|
214
|
+
*/
|
|
215
|
+
const TOOL_ALIASES: Record<string, string> = {
|
|
216
|
+
bash: 'bash_exec',
|
|
217
|
+
shell: 'bash_exec',
|
|
218
|
+
runcommand: 'bash_exec',
|
|
219
|
+
terminal: 'bash_exec',
|
|
220
|
+
read: 'read_file',
|
|
221
|
+
view: 'read_file',
|
|
222
|
+
cat: 'read_file',
|
|
223
|
+
write: 'write_file',
|
|
224
|
+
create: 'write_file',
|
|
225
|
+
edit: 'edit_file',
|
|
226
|
+
strreplace: 'edit_file',
|
|
227
|
+
multiedit: 'edit_file',
|
|
228
|
+
str_replace_editor: 'edit_file',
|
|
229
|
+
search: 'grep',
|
|
230
|
+
ripgrep: 'grep',
|
|
231
|
+
findfiles: 'glob',
|
|
232
|
+
find: 'glob',
|
|
233
|
+
task: 'delegate_task',
|
|
234
|
+
agent: 'delegate_task',
|
|
235
|
+
delegate: 'delegate_task',
|
|
236
|
+
websearch: 'web_search',
|
|
237
|
+
memory: 'recall_memory',
|
|
238
|
+
};
|
|
239
|
+
|
|
240
|
+
/**
|
|
241
|
+
* Reduce a tool name to a comparable form: letters and digits, lower case.
|
|
242
|
+
*
|
|
243
|
+
* @param name - A tool name in any casing or separator style
|
|
244
|
+
* @returns The normalized form, so `GitStatus`, `git_status` and `git-status` all match
|
|
245
|
+
*/
|
|
246
|
+
function normalizeName(name: string): string {
|
|
247
|
+
return name.toLowerCase().replace(/[^a-z0-9]/g, '');
|
|
248
|
+
}
|
|
249
|
+
|
|
250
|
+
/**
|
|
251
|
+
* Match a tool name the model wrote against the tools it actually has.
|
|
252
|
+
*
|
|
253
|
+
* Tried in order: the exact name, the same name in any casing or separator
|
|
254
|
+
* style, then a small table of names other harnesses use. Anything else is
|
|
255
|
+
* unresolved — the caller reports that back rather than guessing, since
|
|
256
|
+
* running the wrong tool is worse than saying the name was wrong.
|
|
257
|
+
*
|
|
258
|
+
* @param written - The name as the model wrote it
|
|
259
|
+
* @param available - The tool names in this run's registry
|
|
260
|
+
* @returns The registry name to call, or undefined when nothing matches
|
|
261
|
+
*
|
|
262
|
+
* @example
|
|
263
|
+
* resolveToolName('Bash', ['bash_exec', 'read_file']) // → 'bash_exec'
|
|
264
|
+
*/
|
|
265
|
+
export function resolveToolName(written: string, available: string[]): string | undefined {
|
|
266
|
+
if (available.includes(written)) return written;
|
|
267
|
+
|
|
268
|
+
const target = normalizeName(written);
|
|
269
|
+
if (!target) return undefined;
|
|
270
|
+
|
|
271
|
+
const byShape = available.find((name) => normalizeName(name) === target);
|
|
272
|
+
if (byShape) return byShape;
|
|
273
|
+
|
|
274
|
+
const aliased = TOOL_ALIASES[target];
|
|
275
|
+
return aliased && available.includes(aliased) ? aliased : undefined;
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
/**
|
|
279
|
+
* Turn the raw string values of a recovered call into arguments the tool accepts.
|
|
280
|
+
*
|
|
281
|
+
* Everything written in text arrives as a string, but a schema may want a
|
|
282
|
+
* number, a boolean or an object. Strings are tried first, so a tool that
|
|
283
|
+
* genuinely wants the string `"42"` still gets it; only if the schema refuses
|
|
284
|
+
* are JSON-looking values parsed.
|
|
285
|
+
*
|
|
286
|
+
* @param raw - Parameter name → raw text value
|
|
287
|
+
* @param schema - The tool's input schema, when it has one
|
|
288
|
+
* @returns The arguments, and the schema's complaint if it still refuses them
|
|
289
|
+
*
|
|
290
|
+
* @example
|
|
291
|
+
* coerceArgs({ timeout: '5000' }, z.object({ timeout: z.number() }))
|
|
292
|
+
* // → { args: { timeout: 5000 } }
|
|
293
|
+
*/
|
|
294
|
+
export function coerceArgs(raw: Record<string, string>, schema?: SchemaLike): CoercedArgs {
|
|
295
|
+
const coerced: Record<string, unknown> = {};
|
|
296
|
+
for (const [key, value] of Object.entries(raw)) {
|
|
297
|
+
const trimmed = value.trim();
|
|
298
|
+
if (JSONISH.test(trimmed)) {
|
|
299
|
+
try {
|
|
300
|
+
coerced[key] = JSON.parse(trimmed);
|
|
301
|
+
continue;
|
|
302
|
+
} catch {
|
|
303
|
+
// Not valid JSON after all — keep the text.
|
|
304
|
+
}
|
|
305
|
+
}
|
|
306
|
+
coerced[key] = value;
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
if (!schema || typeof schema.safeParse !== 'function') return { args: coerced };
|
|
310
|
+
if (schema.safeParse(raw).success) return { args: raw };
|
|
311
|
+
|
|
312
|
+
const second = schema.safeParse(coerced);
|
|
313
|
+
if (second.success) return { args: coerced };
|
|
314
|
+
|
|
315
|
+
return { args: coerced, error: second.error?.message ?? 'arguments did not match the tool schema' };
|
|
316
|
+
}
|
|
@@ -0,0 +1,190 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tests for the runtime half of text-emitted tool-call recovery: executing
|
|
3
|
+
* what the model wrote as prose and feeding the results back into the turn.
|
|
4
|
+
*
|
|
5
|
+
* The bug these lock down: deepseek-chat wrote `<||DSML|| invoke …>` into
|
|
6
|
+
* its reply, the SDK saw no tool call, and the step ended having done
|
|
7
|
+
* nothing — the agent said it would create a team and created nothing
|
|
8
|
+
* (2026-09-19).
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import { describe, it, expect, vi, beforeEach } from 'vitest';
|
|
12
|
+
import { AgentRunnerService } from './agent-runner.service.js';
|
|
13
|
+
import { CREWLY_AGENT_DEFAULTS, type AgentRunResult, type CrewlyAgentConfig } from './types.js';
|
|
14
|
+
|
|
15
|
+
const MAX_STEPS = 50;
|
|
16
|
+
|
|
17
|
+
/** A run result with sensible defaults. */
|
|
18
|
+
function runResult(over: Partial<AgentRunResult> = {}): AgentRunResult {
|
|
19
|
+
return { text: 'text', steps: 1, usage: { input: 10, output: 5 }, toolCalls: [], finishReason: 'stop', ...over };
|
|
20
|
+
}
|
|
21
|
+
|
|
22
|
+
/** The envelope deepseek-chat actually emits, around one call. */
|
|
23
|
+
function dsml(tool: string, param: string, value: string): string {
|
|
24
|
+
return [
|
|
25
|
+
'<||DSML|| calls>',
|
|
26
|
+
`<||DSML|| invoke name="${tool}">`,
|
|
27
|
+
`<||DSML|| parameter name="${param}" string="true">${value}</||DSML|| parameter>`,
|
|
28
|
+
'</||DSML|| invoke>',
|
|
29
|
+
'</||DSML|| calls>',
|
|
30
|
+
].join('\n');
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
describe('salvaging tool calls the model wrote as text', () => {
|
|
34
|
+
let runner: AgentRunnerService;
|
|
35
|
+
let attempts: AgentRunResult[];
|
|
36
|
+
let attemptSpy: ReturnType<typeof vi.fn>;
|
|
37
|
+
let bashExec: ReturnType<typeof vi.fn>;
|
|
38
|
+
let tools: Record<string, unknown>;
|
|
39
|
+
|
|
40
|
+
const config = {
|
|
41
|
+
sessionName: 'test-agent',
|
|
42
|
+
role: 'developer',
|
|
43
|
+
projectPath: '/tmp',
|
|
44
|
+
maxSteps: MAX_STEPS,
|
|
45
|
+
model: { provider: 'deepseek', modelId: 'deepseek-chat' },
|
|
46
|
+
} as unknown as CrewlyAgentConfig;
|
|
47
|
+
|
|
48
|
+
/** Drive the private loop with a scripted sequence of attempts. */
|
|
49
|
+
async function runLoop(): Promise<AgentRunResult> {
|
|
50
|
+
return await (runner as unknown as {
|
|
51
|
+
executeRunWithStreamText(tools: Record<string, unknown>, signal: AbortSignal): Promise<AgentRunResult>;
|
|
52
|
+
}).executeRunWithStreamText(tools, new AbortController().signal);
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
/** The messages the runner has accumulated for the turn. */
|
|
56
|
+
function messages(): Array<{ role: string; content: string }> {
|
|
57
|
+
return (runner as unknown as { state: { messages: Array<{ role: string; content: string }> } }).state.messages;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
beforeEach(() => {
|
|
61
|
+
runner = new AgentRunnerService(config);
|
|
62
|
+
attempts = [];
|
|
63
|
+
attemptSpy = vi.fn(async () => {
|
|
64
|
+
const next = attempts.shift() ?? runResult();
|
|
65
|
+
// Mirror what a real attempt does: its text lands in the transcript.
|
|
66
|
+
if (next.text) messages().push({ role: 'assistant', content: next.text });
|
|
67
|
+
return next;
|
|
68
|
+
});
|
|
69
|
+
(runner as unknown as Record<string, unknown>).attemptWithErrorRetries = attemptSpy;
|
|
70
|
+
bashExec = vi.fn(async () => ({ success: true, stdout: 'M src/index.ts', exitCode: 0 }));
|
|
71
|
+
tools = { bash_exec: { description: 'run a command', inputSchema: undefined, execute: bashExec } };
|
|
72
|
+
vi.spyOn(console, 'warn').mockImplementation(() => undefined);
|
|
73
|
+
});
|
|
74
|
+
|
|
75
|
+
it('executes the call, hides the markup, and lets the turn finish', async () => {
|
|
76
|
+
attempts = [
|
|
77
|
+
runResult({ text: `Checking the repo.\n${dsml('bash_exec', 'command', 'git status --short')}` }),
|
|
78
|
+
runResult({ text: 'One file is modified.' }),
|
|
79
|
+
];
|
|
80
|
+
const out = await runLoop();
|
|
81
|
+
|
|
82
|
+
expect(bashExec).toHaveBeenCalledWith({ command: 'git status --short' });
|
|
83
|
+
expect(out.text).toBe('Checking the repo.\n\nOne file is modified.');
|
|
84
|
+
expect(out.text).not.toContain('DSML');
|
|
85
|
+
expect(out.incomplete).toBeUndefined();
|
|
86
|
+
});
|
|
87
|
+
|
|
88
|
+
it('records the salvaged call in the run ledger, so it is not invisible work', async () => {
|
|
89
|
+
attempts = [runResult({ text: dsml('bash_exec', 'command', 'ls') }), runResult({ text: 'done' })];
|
|
90
|
+
const out = await runLoop();
|
|
91
|
+
expect(out.toolCalls).toEqual([
|
|
92
|
+
{ toolName: 'bash_exec', args: { command: 'ls' }, result: { success: true, stdout: 'M src/index.ts', exitCode: 0 } },
|
|
93
|
+
]);
|
|
94
|
+
});
|
|
95
|
+
|
|
96
|
+
it('feeds the result back and tells the model to use real tool calls', async () => {
|
|
97
|
+
attempts = [runResult({ text: dsml('bash_exec', 'command', 'ls') }), runResult({ text: 'done' })];
|
|
98
|
+
await runLoop();
|
|
99
|
+
const fedBack = messages().filter((m) => m.role === 'user').pop();
|
|
100
|
+
expect(fedBack?.content).toContain('M src/index.ts');
|
|
101
|
+
expect(fedBack?.content).toMatch(/executes nothing/i);
|
|
102
|
+
});
|
|
103
|
+
|
|
104
|
+
it('rewrites the transcript so the markup is not modelled as the way to call a tool', async () => {
|
|
105
|
+
attempts = [runResult({ text: `Checking.\n${dsml('bash_exec', 'command', 'ls')}` }), runResult({ text: 'done' })];
|
|
106
|
+
await runLoop();
|
|
107
|
+
expect(messages().some((m) => m.content.includes('DSML'))).toBe(false);
|
|
108
|
+
expect(messages().some((m) => m.role === 'assistant' && m.content === 'Checking.')).toBe(true);
|
|
109
|
+
});
|
|
110
|
+
|
|
111
|
+
it("runs the tool under the name another harness uses — the live failure was 'Bash'", async () => {
|
|
112
|
+
attempts = [runResult({ text: dsml('Bash', 'command', 'ls') }), runResult({ text: 'ok' })];
|
|
113
|
+
await runLoop();
|
|
114
|
+
expect(bashExec).toHaveBeenCalledWith({ command: 'ls' });
|
|
115
|
+
const fedBack = messages().filter((m) => m.role === 'user').pop();
|
|
116
|
+
expect(fedBack?.content).toContain('bash_exec (you wrote "Bash")');
|
|
117
|
+
});
|
|
118
|
+
|
|
119
|
+
it('reports a name it cannot resolve instead of guessing at a tool', async () => {
|
|
120
|
+
attempts = [runResult({ text: dsml('deploy_to_prod', 'target', 'live') }), runResult({ text: 'ok' })];
|
|
121
|
+
await runLoop();
|
|
122
|
+
expect(bashExec).not.toHaveBeenCalled();
|
|
123
|
+
const fedBack = messages().filter((m) => m.role === 'user').pop();
|
|
124
|
+
expect(fedBack?.content).toMatch(/no tool with that name/i);
|
|
125
|
+
expect(fedBack?.content).toContain('bash_exec');
|
|
126
|
+
});
|
|
127
|
+
|
|
128
|
+
it('reports a schema complaint rather than calling the tool with bad arguments', async () => {
|
|
129
|
+
tools = {
|
|
130
|
+
bash_exec: {
|
|
131
|
+
execute: bashExec,
|
|
132
|
+
inputSchema: { safeParse: () => ({ success: false, error: { message: 'command is required' } }) },
|
|
133
|
+
},
|
|
134
|
+
};
|
|
135
|
+
attempts = [runResult({ text: dsml('bash_exec', 'cmd', 'ls') }), runResult({ text: 'ok' })];
|
|
136
|
+
await runLoop();
|
|
137
|
+
expect(bashExec).not.toHaveBeenCalled();
|
|
138
|
+
expect(messages().filter((m) => m.role === 'user').pop()?.content).toContain('command is required');
|
|
139
|
+
});
|
|
140
|
+
|
|
141
|
+
it('hands a thrown tool error back as a result instead of losing the turn', async () => {
|
|
142
|
+
bashExec.mockRejectedValueOnce(new Error('spawn ENOENT'));
|
|
143
|
+
attempts = [runResult({ text: dsml('bash_exec', 'command', 'nope') }), runResult({ text: 'ok' })];
|
|
144
|
+
const out = await runLoop();
|
|
145
|
+
expect(out.toolCalls[0].result).toEqual({ error: 'spawn ENOENT' });
|
|
146
|
+
expect(messages().filter((m) => m.role === 'user').pop()?.content).toContain('spawn ENOENT');
|
|
147
|
+
});
|
|
148
|
+
|
|
149
|
+
it('gives up after the salvage budget rather than looping on a model that will not learn', async () => {
|
|
150
|
+
attempts = Array.from({ length: 12 }, () => runResult({ text: dsml('bash_exec', 'command', 'ls') }));
|
|
151
|
+
await runLoop();
|
|
152
|
+
expect(bashExec).toHaveBeenCalledTimes(CREWLY_AGENT_DEFAULTS.MAX_TEXT_TOOL_SALVAGES);
|
|
153
|
+
});
|
|
154
|
+
|
|
155
|
+
it('caps how many calls one confused turn can fan out to', async () => {
|
|
156
|
+
const many = Array.from({ length: 9 }, (_, i) => `<invoke name="bash_exec"><parameter name="command">echo ${i}</parameter></invoke>`).join('\n');
|
|
157
|
+
attempts = [runResult({ text: many }), runResult({ text: 'ok' })];
|
|
158
|
+
await runLoop();
|
|
159
|
+
expect(bashExec).toHaveBeenCalledTimes(CREWLY_AGENT_DEFAULTS.MAX_SALVAGED_CALLS_PER_ROUND);
|
|
160
|
+
expect(messages().filter((m) => m.role === 'user').pop()?.content).toMatch(/4 further call\(s\) were not run/);
|
|
161
|
+
});
|
|
162
|
+
|
|
163
|
+
it('truncates a huge result so one salvaged command cannot blow the context', async () => {
|
|
164
|
+
bashExec.mockResolvedValueOnce('x'.repeat(CREWLY_AGENT_DEFAULTS.SALVAGED_RESULT_MAX_CHARS + 500));
|
|
165
|
+
attempts = [runResult({ text: dsml('bash_exec', 'command', 'cat big') }), runResult({ text: 'ok' })];
|
|
166
|
+
await runLoop();
|
|
167
|
+
const fedBack = messages().filter((m) => m.role === 'user').pop()?.content ?? '';
|
|
168
|
+
expect(fedBack).toMatch(/truncated, 500 more characters/);
|
|
169
|
+
expect(fedBack.length).toBeLessThan(CREWLY_AGENT_DEFAULTS.SALVAGED_RESULT_MAX_CHARS + 1500);
|
|
170
|
+
});
|
|
171
|
+
|
|
172
|
+
it('leaves a healthy turn completely alone', async () => {
|
|
173
|
+
attempts = [runResult({ text: 'All done, nothing to salvage.' })];
|
|
174
|
+
const out = await runLoop();
|
|
175
|
+
expect(attemptSpy).toHaveBeenCalledTimes(1);
|
|
176
|
+
expect(bashExec).not.toHaveBeenCalled();
|
|
177
|
+
expect(out.text).toBe('All done, nothing to salvage.');
|
|
178
|
+
});
|
|
179
|
+
|
|
180
|
+
it('still reports the turn incomplete when the retried attempt bails', async () => {
|
|
181
|
+
attempts = [
|
|
182
|
+
runResult({ text: dsml('bash_exec', 'command', 'ls') }),
|
|
183
|
+
runResult({ text: '', finishReason: 'other' }),
|
|
184
|
+
runResult({ text: '', finishReason: 'other' }),
|
|
185
|
+
];
|
|
186
|
+
const out = await runLoop();
|
|
187
|
+
expect(bashExec).toHaveBeenCalledTimes(1);
|
|
188
|
+
expect(out.incomplete).toMatchObject({ reason: 'abnormal-finish' });
|
|
189
|
+
});
|
|
190
|
+
});
|
|
@@ -556,6 +556,12 @@ export const CREWLY_AGENT_DEFAULTS = {
|
|
|
556
556
|
MAX_CONTINUATIONS: 3,
|
|
557
557
|
/** Re-runs allowed after the provider ended a turn abnormally. */
|
|
558
558
|
MAX_ABNORMAL_RETRIES: 1,
|
|
559
|
+
/** Rounds allowed for executing tool calls a model wrote as text instead of calling. */
|
|
560
|
+
MAX_TEXT_TOOL_SALVAGES: 3,
|
|
561
|
+
/** Tool calls executed per salvage round, so one confused turn cannot fan out. */
|
|
562
|
+
MAX_SALVAGED_CALLS_PER_ROUND: 5,
|
|
563
|
+
/** Characters of a salvaged tool result fed back to the model. */
|
|
564
|
+
SALVAGED_RESULT_MAX_CHARS: 4000,
|
|
559
565
|
/** Maximum tool calls allowed per single response to prevent polling dead-loops */
|
|
560
566
|
MAX_TOOL_CALLS_PER_RESPONSE: 15,
|
|
561
567
|
/** Consecutive identical tool calls before aborting (loop detection) */
|