crewly 1.20.34 → 1.20.40

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (78) hide show
  1. package/config/skills/_common/lib.sh +6 -0
  2. package/config/skills/agent/core/calendar-create/SKILL.md +10 -0
  3. package/config/skills/agent/core/calendar-create/execute.sh +6 -0
  4. package/config/skills/agent/core/calendar-list/SKILL.md +10 -0
  5. package/config/skills/agent/core/calendar-list/execute.sh +6 -0
  6. package/config/skills/agent/core/docs-read/SKILL.md +10 -0
  7. package/config/skills/agent/core/docs-read/execute.sh +6 -1
  8. package/config/skills/agent/core/docs-write/SKILL.md +10 -0
  9. package/config/skills/agent/core/docs-write/execute.sh +6 -1
  10. package/config/skills/agent/core/drive-read/SKILL.md +10 -0
  11. package/config/skills/agent/core/drive-read/execute.sh +6 -1
  12. package/config/skills/agent/core/drive-search/SKILL.md +10 -0
  13. package/config/skills/agent/core/drive-search/execute.sh +6 -1
  14. package/config/skills/agent/core/drive-upload/SKILL.md +10 -0
  15. package/config/skills/agent/core/drive-upload/execute.sh +6 -1
  16. package/config/skills/agent/core/gmail-read/SKILL.md +10 -0
  17. package/config/skills/agent/core/gmail-read/execute.sh +6 -0
  18. package/config/skills/agent/core/gmail-search/SKILL.md +10 -0
  19. package/config/skills/agent/core/gmail-search/execute.sh +6 -0
  20. package/config/skills/agent/core/gmail-send/SKILL.md +10 -0
  21. package/config/skills/agent/core/gmail-send/execute.sh +6 -0
  22. package/config/skills/agent/core/sheets-read/SKILL.md +10 -0
  23. package/config/skills/agent/core/sheets-read/execute.sh +6 -1
  24. package/config/skills/agent/core/sheets-write/SKILL.md +10 -0
  25. package/config/skills/agent/core/sheets-write/execute.sh +6 -1
  26. package/config/skills/agent/core/slides-create/SKILL.md +10 -0
  27. package/config/skills/agent/core/slides-create/execute.sh +6 -1
  28. package/config/skills/agent/core/slides-read/SKILL.md +10 -0
  29. package/config/skills/agent/core/slides-read/execute.sh +6 -1
  30. package/config/slack-app-manifest.json +16 -9
  31. package/dist/backend/backend/src/constants.d.ts +18 -4
  32. package/dist/backend/backend/src/constants.d.ts.map +1 -1
  33. package/dist/backend/backend/src/constants.js +16 -4
  34. package/dist/backend/backend/src/constants.js.map +1 -1
  35. package/dist/backend/backend/src/controllers/google/google.controller.d.ts +8 -0
  36. package/dist/backend/backend/src/controllers/google/google.controller.d.ts.map +1 -1
  37. package/dist/backend/backend/src/controllers/google/google.controller.js +137 -37
  38. package/dist/backend/backend/src/controllers/google/google.controller.js.map +1 -1
  39. package/dist/backend/backend/src/controllers/google/google.routes.d.ts +2 -1
  40. package/dist/backend/backend/src/controllers/google/google.routes.d.ts.map +1 -1
  41. package/dist/backend/backend/src/controllers/google/google.routes.js +4 -2
  42. package/dist/backend/backend/src/controllers/google/google.routes.js.map +1 -1
  43. package/dist/backend/backend/src/controllers/slack/slack-error.utils.d.ts +46 -0
  44. package/dist/backend/backend/src/controllers/slack/slack-error.utils.d.ts.map +1 -0
  45. package/dist/backend/backend/src/controllers/slack/slack-error.utils.js +54 -0
  46. package/dist/backend/backend/src/controllers/slack/slack-error.utils.js.map +1 -0
  47. package/dist/backend/backend/src/controllers/slack/slack.controller.d.ts.map +1 -1
  48. package/dist/backend/backend/src/controllers/slack/slack.controller.js +5 -12
  49. package/dist/backend/backend/src/controllers/slack/slack.controller.js.map +1 -1
  50. package/dist/backend/backend/src/services/google/google-api.client.d.ts +23 -2
  51. package/dist/backend/backend/src/services/google/google-api.client.d.ts.map +1 -1
  52. package/dist/backend/backend/src/services/google/google-api.client.js +5 -2
  53. package/dist/backend/backend/src/services/google/google-api.client.js.map +1 -1
  54. package/dist/backend/backend/src/services/google/google-workspace-token.service.d.ts +61 -11
  55. package/dist/backend/backend/src/services/google/google-workspace-token.service.d.ts.map +1 -1
  56. package/dist/backend/backend/src/services/google/google-workspace-token.service.js +108 -31
  57. package/dist/backend/backend/src/services/google/google-workspace-token.service.js.map +1 -1
  58. package/dist/backend/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
  59. package/dist/backend/backend/src/services/slack/slack-team-channel.service.js +27 -4
  60. package/dist/backend/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
  61. package/dist/backend/build-info.json +2 -2
  62. package/dist/cli/backend/src/constants.d.ts +18 -4
  63. package/dist/cli/backend/src/constants.d.ts.map +1 -1
  64. package/dist/cli/backend/src/constants.js +16 -4
  65. package/dist/cli/backend/src/constants.js.map +1 -1
  66. package/dist/cli/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
  67. package/dist/cli/backend/src/services/slack/slack-team-channel.service.js +27 -4
  68. package/dist/cli/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
  69. package/frontend/dist/assets/{index-e079a375.js → index-e7785269.js} +267 -267
  70. package/frontend/dist/index.html +1 -1
  71. package/package.json +1 -1
  72. package/packages/crewly-agent/src/runtime/agent-runner.service.test.ts +10 -1
  73. package/packages/crewly-agent/src/runtime/agent-runner.service.ts +321 -1
  74. package/packages/crewly-agent/src/runtime/finish-recovery.test.ts +177 -0
  75. package/packages/crewly-agent/src/runtime/text-tool-calls.test.ts +144 -0
  76. package/packages/crewly-agent/src/runtime/text-tool-calls.ts +316 -0
  77. package/packages/crewly-agent/src/runtime/text-tool-salvage.test.ts +190 -0
  78. package/packages/crewly-agent/src/runtime/types.ts +38 -0
@@ -0,0 +1,144 @@
1
+ /**
2
+ * Tests for recovering tool calls a model wrote as text.
3
+ *
4
+ * The bug these lock down: deepseek-chat writes its call envelope into the
5
+ * text channel with its own separators (`<||DSML|| invoke name="Bash">`),
6
+ * so no tool ever ran and the user was shown the markup instead of an answer
7
+ * (2026-09-19). Matching is by shape — a new prefix must not defeat it.
8
+ */
9
+
10
+ import { describe, it, expect } from 'vitest';
11
+ import { parseTextToolCalls, hasTextToolCalls, coerceArgs, resolveToolName, type SchemaLike } from './text-tool-calls.js';
12
+
13
+ /** The exact bytes deepseek-chat emitted, taken from ~/.crewly/chat.db. */
14
+ const DSML = [
15
+ 'Let me check the repo.',
16
+ '<||DSML|| calls>',
17
+ '<||DSML|| invoke name="bash_exec">',
18
+ '<||DSML|| parameter name="command" string="true">git status --short</||DSML|| parameter>',
19
+ '<||DSML|| parameter name="timeout" string="true">5000</||DSML|| parameter>',
20
+ '</||DSML|| invoke>',
21
+ '</||DSML|| calls>',
22
+ ].join('\n');
23
+
24
+ /** A minimal Zod-style schema. */
25
+ function schema(check: (v: Record<string, unknown>) => boolean): SchemaLike {
26
+ return { safeParse: (v: unknown) => (check(v as Record<string, unknown>) ? { success: true, data: v } : { success: false, error: { message: 'bad args' } }) };
27
+ }
28
+
29
+ describe('parseTextToolCalls', () => {
30
+ it('recovers the deepseek envelope, prefix and all, and leaves the prose', () => {
31
+ const { calls, text } = parseTextToolCalls(DSML);
32
+ expect(calls).toEqual([
33
+ { toolName: 'bash_exec', args: { command: 'git status --short', timeout: '5000' } },
34
+ ]);
35
+ expect(text).toBe('Let me check the repo.');
36
+ });
37
+
38
+ it('recovers the plain Claude-style envelope too', () => {
39
+ const { calls } = parseTextToolCalls('<function_calls><invoke name="read_file"><parameter name="path">a.ts</parameter></invoke></function_calls>');
40
+ expect(calls).toEqual([{ toolName: 'read_file', args: { path: 'a.ts' } }]);
41
+ });
42
+
43
+ it('recovers every call in a batch, in order', () => {
44
+ const { calls } = parseTextToolCalls(
45
+ '<invoke name="a"><parameter name="x">1</parameter></invoke><invoke name="b"><parameter name="y">2</parameter></invoke>',
46
+ );
47
+ expect(calls.map((c) => c.toolName)).toEqual(['a', 'b']);
48
+ expect(calls[1].args).toEqual({ y: '2' });
49
+ });
50
+
51
+ it('keeps a multi-line value intact, including its own angle brackets', () => {
52
+ const { calls } = parseTextToolCalls(
53
+ '<invoke name="write_file"><parameter name="content">line 1\nif (a < b) { go(); }\nline 3</parameter></invoke>',
54
+ );
55
+ expect(calls[0].args.content).toBe('line 1\nif (a < b) { go(); }\nline 3');
56
+ });
57
+
58
+ it('salvages a call the model never closed', () => {
59
+ const { calls, text } = parseTextToolCalls('Working.\n<invoke name="bash_exec"><parameter name="command">npm test');
60
+ expect(calls).toEqual([{ toolName: 'bash_exec', args: { command: 'npm test' } }]);
61
+ expect(text).toBe('Working.');
62
+ });
63
+
64
+ it('leaves a fenced example alone — it is documentation, not a call', () => {
65
+ const raw = 'Here is the syntax:\n\n```\n<invoke name="bash_exec"><parameter name="command">ls</parameter></invoke>\n```';
66
+ const { calls, text } = parseTextToolCalls(raw);
67
+ expect(calls).toEqual([]);
68
+ expect(text).toBe(raw);
69
+ });
70
+
71
+ it('finds nothing in ordinary prose, and does not touch it', () => {
72
+ for (const prose of ['', 'The build recalls the cached layer.', 'Use <Parameter> in the docs? no.']) {
73
+ expect(parseTextToolCalls(prose).calls).toEqual([]);
74
+ }
75
+ expect(parseTextToolCalls('The build recalls the cached layer.').text).toBe('The build recalls the cached layer.');
76
+ });
77
+
78
+ it('ignores an invoke tag with no name rather than inventing one', () => {
79
+ expect(parseTextToolCalls('<invoke><parameter name="x">1</parameter></invoke>').calls).toEqual([]);
80
+ });
81
+ });
82
+
83
+ describe('hasTextToolCalls', () => {
84
+ it('is true for any prefix and false for prose', () => {
85
+ expect(hasTextToolCalls(DSML)).toBe(true);
86
+ expect(hasTextToolCalls('<invoke name="x">')).toBe(true);
87
+ expect(hasTextToolCalls('nothing to see')).toBe(false);
88
+ });
89
+ });
90
+
91
+ describe('coerceArgs', () => {
92
+ it('passes strings through when the schema accepts them', () => {
93
+ const out = coerceArgs({ command: '5' }, schema((v) => typeof v.command === 'string'));
94
+ expect(out).toEqual({ args: { command: '5' } });
95
+ });
96
+
97
+ it('parses JSON-looking values only when the schema refuses the strings', () => {
98
+ const out = coerceArgs({ timeout: '5000', on: 'true', tags: '["a"]' }, schema((v) => typeof v.timeout === 'number'));
99
+ expect(out.args).toEqual({ timeout: 5000, on: true, tags: ['a'] });
100
+ expect(out.error).toBeUndefined();
101
+ });
102
+
103
+ it('coerces with no schema to give the tool its best shot', () => {
104
+ expect(coerceArgs({ n: '3', s: 'hello' }).args).toEqual({ n: 3, s: 'hello' });
105
+ });
106
+
107
+ it('reports the schema complaint instead of calling a tool with bad arguments', () => {
108
+ const out = coerceArgs({ command: 'ls' }, schema((v) => typeof v.path === 'string'));
109
+ expect(out.error).toBe('bad args');
110
+ });
111
+
112
+ it('keeps a value that only looks like JSON', () => {
113
+ expect(coerceArgs({ command: '{ not json' }).args).toEqual({ command: '{ not json' });
114
+ });
115
+ });
116
+
117
+ describe('resolveToolName', () => {
118
+ const available = ['bash_exec', 'read_file', 'write_file', 'edit_file', 'glob', 'grep', 'delegate_task', 'git_status'];
119
+
120
+ it('takes an exact name unchanged', () => {
121
+ expect(resolveToolName('bash_exec', available)).toBe('bash_exec');
122
+ });
123
+
124
+ it('matches across casing and separators, so GitStatus finds git_status', () => {
125
+ for (const written of ['GitStatus', 'git-status', 'GIT_STATUS', 'gitStatus']) {
126
+ expect(resolveToolName(written, available)).toBe('git_status');
127
+ }
128
+ });
129
+
130
+ it("resolves the names another harness uses — the live failure was 'Bash'", () => {
131
+ expect(resolveToolName('Bash', available)).toBe('bash_exec');
132
+ expect(resolveToolName('Read', available)).toBe('read_file');
133
+ expect(resolveToolName('Write', available)).toBe('write_file');
134
+ expect(resolveToolName('Edit', available)).toBe('edit_file');
135
+ expect(resolveToolName('Task', available)).toBe('delegate_task');
136
+ expect(resolveToolName('Search', available)).toBe('grep');
137
+ });
138
+
139
+ it('does not invent a tool the run does not have', () => {
140
+ expect(resolveToolName('Bash', ['read_file'])).toBeUndefined();
141
+ expect(resolveToolName('deploy_to_prod', available)).toBeUndefined();
142
+ expect(resolveToolName('', available)).toBeUndefined();
143
+ });
144
+ });
@@ -0,0 +1,316 @@
1
+ /**
2
+ * Recover tool calls a model wrote as text.
3
+ *
4
+ * A weak model sometimes *writes* its function-call envelope into the text
5
+ * channel instead of emitting a real tool call. The provider sees ordinary
6
+ * prose, the AI SDK sees no tool call, the step ends — so nothing runs, and
7
+ * the user is shown a wall of markup where an answer should be. deepseek-chat
8
+ * does this with its own separators:
9
+ *
10
+ * `<||DSML|| invoke name="Bash">` … `<||DSML|| parameter name="command" …>`
11
+ *
12
+ * which is Claude's `invoke`/`parameter` grammar with `||DSML|| ` (U+FF5C
13
+ * pipes) where the namespace prefix goes. Nothing in Crewly teaches that
14
+ * format — the model emits it on its own (2026-09-19, #crewly-support).
15
+ *
16
+ * Stripping the markup hides the symptom and still loses the work. This
17
+ * module instead *parses* the envelope so the runtime can execute what the
18
+ * model meant to call and hand the results back, turning a dead turn into a
19
+ * working one. Matching is by **shape**, never by a list of known names: the
20
+ * prefix is whatever the model felt like emitting.
21
+ *
22
+ * @module runtime/text-tool-calls
23
+ */
24
+
25
+ /** One tool call recovered from text. Values are raw, as written. */
26
+ export interface TextToolCall {
27
+ /** Tool name from the `name="…"` attribute. */
28
+ toolName: string;
29
+ /** Parameter name → raw text value. */
30
+ args: Record<string, string>;
31
+ }
32
+
33
+ /** Result of {@link parseTextToolCalls}. */
34
+ export interface ParsedTextToolCalls {
35
+ /** The calls found, in the order they were written. */
36
+ calls: TextToolCall[];
37
+ /** The same text with every envelope removed and whitespace tidied. */
38
+ text: string;
39
+ }
40
+
41
+ /** Anything with a Zod-style `safeParse` — kept duck-typed so this module stays dependency-free. */
42
+ export interface SchemaLike {
43
+ safeParse(value: unknown): { success: boolean; data?: unknown; error?: { message?: string } };
44
+ }
45
+
46
+ /** Result of {@link coerceArgs}. */
47
+ export interface CoercedArgs {
48
+ /** Arguments to pass to the tool. */
49
+ args: Record<string, unknown>;
50
+ /** Why the schema rejected them, when it did. */
51
+ error?: string;
52
+ }
53
+
54
+ /**
55
+ * A tag like `<invoke …>`, `</invoke>`, `<||DSML|| parameter …>` — any
56
+ * junk or namespace before the name, anything but `<`/`>` after it.
57
+ *
58
+ * @param name - Tag-name pattern
59
+ * @param closing - Match the closing form
60
+ * @returns The pattern source
61
+ */
62
+ function tagSource(name: string, closing: boolean): string {
63
+ return `<${closing ? '\\s*/' : ''}[^<>]{0,40}?${name}\\b[^<>]*>`;
64
+ }
65
+
66
+ /** Wrapper the model puts around a batch of calls: `function_calls`, `tool_calls`, or a bare `calls`. */
67
+ const WRAPPER_NAME = '(?:function_calls|tool_calls|(?<![A-Za-z_])calls)';
68
+
69
+ const INVOKE_OPEN = new RegExp(tagSource('invoke', false), 'gi');
70
+ const INVOKE_CLOSE = new RegExp(tagSource('invoke', true), 'i');
71
+ const PARAM_OPEN = new RegExp(tagSource('parameter', false), 'gi');
72
+ const PARAM_CLOSE = new RegExp(tagSource('parameter', true), 'i');
73
+
74
+ /** `name="x"` / `name='x'` / `name=x` inside a tag. */
75
+ const NAME_ATTR = /\bname\s*=\s*(?:"([^"]*)"|'([^']*)'|([^\s"'<>]+))/i;
76
+
77
+ /** Well-formed `<x>…</x>` blocks, for removal. */
78
+ const BLOCK_PATTERNS = [WRAPPER_NAME, 'invoke'].map(
79
+ (name) => new RegExp(`${tagSource(name, false)}[\\s\\S]*?${tagSource(name, true)}`, 'gi'),
80
+ );
81
+
82
+ /** An opening tag with no close — the output was cut off mid-call. */
83
+ const UNCLOSED_PATTERN = new RegExp(`(?:${tagSource(WRAPPER_NAME, false)}|${tagSource('invoke', false)})[\\s\\S]*$`, 'i');
84
+
85
+ /** Any leftover lone tag. */
86
+ const LONE_TAG_PATTERN = new RegExp(
87
+ [WRAPPER_NAME, 'invoke', 'parameter'].map((name) => `${tagSource(name, false)}|${tagSource(name, true)}`).join('|'),
88
+ 'gi',
89
+ );
90
+
91
+ /** A value that is plainly JSON rather than prose. */
92
+ const JSONISH = /^(?:true|false|null|-?\d+(?:\.\d+)?(?:[eE][+-]?\d+)?|\{[\s\S]*\}|\[[\s\S]*\])$/;
93
+
94
+ /**
95
+ * Read the `name="…"` attribute out of an opening tag.
96
+ *
97
+ * @param tag - The whole tag, angle brackets included
98
+ * @returns The name, or '' when the tag has none
99
+ */
100
+ function nameOf(tag: string): string {
101
+ const m = NAME_ATTR.exec(tag);
102
+ return (m?.[1] ?? m?.[2] ?? m?.[3] ?? '').trim();
103
+ }
104
+
105
+ /**
106
+ * Split text into segments, marking the fenced ones.
107
+ *
108
+ * Markup inside a ``` fence is an example the agent meant to show — it is
109
+ * neither executed nor stripped.
110
+ *
111
+ * @param input - Raw text
112
+ * @returns Segments; odd indexes are fenced code
113
+ */
114
+ function splitOnFences(input: string): string[] {
115
+ return input.split(/(```[\s\S]*?```)/g);
116
+ }
117
+
118
+ /**
119
+ * Parse the `parameter` blocks inside one `invoke` body.
120
+ *
121
+ * A block runs to its closing tag, or — when the model never wrote one — to
122
+ * the next parameter or the end of the body, so a truncated call still yields
123
+ * the arguments it managed to write.
124
+ *
125
+ * @param body - Text between the invoke tags
126
+ * @returns Parameter name → raw value
127
+ */
128
+ function parseParams(body: string): Record<string, string> {
129
+ const args: Record<string, string> = {};
130
+ PARAM_OPEN.lastIndex = 0;
131
+ const opens: Array<{ name: string; from: number }> = [];
132
+ for (let m = PARAM_OPEN.exec(body); m; m = PARAM_OPEN.exec(body)) {
133
+ opens.push({ name: nameOf(m[0]), from: m.index + m[0].length });
134
+ }
135
+ opens.forEach((open, i) => {
136
+ const until = i + 1 < opens.length ? body.lastIndexOf('<', opens[i + 1].from) : body.length;
137
+ const slice = body.slice(open.from, Math.max(until, open.from));
138
+ const close = PARAM_CLOSE.exec(slice);
139
+ const value = close ? slice.slice(0, close.index) : slice;
140
+ if (open.name) args[open.name] = value.replace(/^\r?\n/, '').replace(/\s+$/, '');
141
+ });
142
+ return args;
143
+ }
144
+
145
+ /**
146
+ * Find every tool call written as text, and return the text without them.
147
+ *
148
+ * @param raw - The model's text output
149
+ * @returns The recovered calls and the cleaned text
150
+ *
151
+ * @example
152
+ * parseTextToolCalls('<invoke name="bash_exec"><parameter name="command">ls</parameter></invoke>')
153
+ * // → { calls: [{ toolName: 'bash_exec', args: { command: 'ls' } }], text: '' }
154
+ */
155
+ export function parseTextToolCalls(raw: string): ParsedTextToolCalls {
156
+ const input = raw ?? '';
157
+ if (!input) return { calls: [], text: '' };
158
+
159
+ const calls: TextToolCall[] = [];
160
+ const cleaned = splitOnFences(input)
161
+ .map((segment, i) => {
162
+ if (i % 2 === 1) return segment; // fenced code — an example, not a call
163
+
164
+ INVOKE_OPEN.lastIndex = 0;
165
+ const opens: Array<{ name: string; from: number }> = [];
166
+ for (let m = INVOKE_OPEN.exec(segment); m; m = INVOKE_OPEN.exec(segment)) {
167
+ opens.push({ name: nameOf(m[0]), from: m.index + m[0].length });
168
+ }
169
+ opens.forEach((open, idx) => {
170
+ // An unclosed invoke runs to the next one, or to the end of the text.
171
+ const until = idx + 1 < opens.length ? segment.lastIndexOf('<', opens[idx + 1].from) : segment.length;
172
+ const slice = segment.slice(open.from, Math.max(until, open.from));
173
+ const close = INVOKE_CLOSE.exec(slice);
174
+ const body = close ? slice.slice(0, close.index) : slice;
175
+ if (open.name) calls.push({ toolName: open.name, args: parseParams(body) });
176
+ });
177
+
178
+ let out = segment;
179
+ for (const pattern of BLOCK_PATTERNS) out = out.replace(pattern, '');
180
+ out = out.replace(UNCLOSED_PATTERN, '');
181
+ return out.replace(LONE_TAG_PATTERN, '');
182
+ })
183
+ .join('');
184
+
185
+ const text = cleaned
186
+ .split('\n')
187
+ .map((line) => line.replace(/[ \t]+$/, ''))
188
+ .join('\n')
189
+ .replace(/\n{3,}/g, '\n\n')
190
+ .trim();
191
+
192
+ return { calls, text };
193
+ }
194
+
195
+ /**
196
+ * Whether text contains something that looks like a tool-call envelope.
197
+ *
198
+ * @param raw - Candidate text
199
+ * @returns True when an `invoke` or wrapper tag is present
200
+ */
201
+ export function hasTextToolCalls(raw: string): boolean {
202
+ return new RegExp(`${tagSource(WRAPPER_NAME, false)}|${tagSource('invoke', false)}`, 'i').test(raw ?? '');
203
+ }
204
+
205
+ /**
206
+ * Names other harnesses use for tools Crewly spells differently.
207
+ *
208
+ * A model that has seen Claude Code reaches for `Bash` and `Read`; ours are
209
+ * `bash_exec` and `read_file`. Observed live on the first turn after the
210
+ * salvage shipped: deepseek-chat wrote `Bash` twice, was told the name did
211
+ * not exist, and only then used real names (2026-09-19). Resolving the alias
212
+ * turns that wasted round into a working one. This maps onto tools the agent
213
+ * already has — it never grants one it does not.
214
+ */
215
+ const TOOL_ALIASES: Record<string, string> = {
216
+ bash: 'bash_exec',
217
+ shell: 'bash_exec',
218
+ runcommand: 'bash_exec',
219
+ terminal: 'bash_exec',
220
+ read: 'read_file',
221
+ view: 'read_file',
222
+ cat: 'read_file',
223
+ write: 'write_file',
224
+ create: 'write_file',
225
+ edit: 'edit_file',
226
+ strreplace: 'edit_file',
227
+ multiedit: 'edit_file',
228
+ str_replace_editor: 'edit_file',
229
+ search: 'grep',
230
+ ripgrep: 'grep',
231
+ findfiles: 'glob',
232
+ find: 'glob',
233
+ task: 'delegate_task',
234
+ agent: 'delegate_task',
235
+ delegate: 'delegate_task',
236
+ websearch: 'web_search',
237
+ memory: 'recall_memory',
238
+ };
239
+
240
+ /**
241
+ * Reduce a tool name to a comparable form: letters and digits, lower case.
242
+ *
243
+ * @param name - A tool name in any casing or separator style
244
+ * @returns The normalized form, so `GitStatus`, `git_status` and `git-status` all match
245
+ */
246
+ function normalizeName(name: string): string {
247
+ return name.toLowerCase().replace(/[^a-z0-9]/g, '');
248
+ }
249
+
250
+ /**
251
+ * Match a tool name the model wrote against the tools it actually has.
252
+ *
253
+ * Tried in order: the exact name, the same name in any casing or separator
254
+ * style, then a small table of names other harnesses use. Anything else is
255
+ * unresolved — the caller reports that back rather than guessing, since
256
+ * running the wrong tool is worse than saying the name was wrong.
257
+ *
258
+ * @param written - The name as the model wrote it
259
+ * @param available - The tool names in this run's registry
260
+ * @returns The registry name to call, or undefined when nothing matches
261
+ *
262
+ * @example
263
+ * resolveToolName('Bash', ['bash_exec', 'read_file']) // → 'bash_exec'
264
+ */
265
+ export function resolveToolName(written: string, available: string[]): string | undefined {
266
+ if (available.includes(written)) return written;
267
+
268
+ const target = normalizeName(written);
269
+ if (!target) return undefined;
270
+
271
+ const byShape = available.find((name) => normalizeName(name) === target);
272
+ if (byShape) return byShape;
273
+
274
+ const aliased = TOOL_ALIASES[target];
275
+ return aliased && available.includes(aliased) ? aliased : undefined;
276
+ }
277
+
278
+ /**
279
+ * Turn the raw string values of a recovered call into arguments the tool accepts.
280
+ *
281
+ * Everything written in text arrives as a string, but a schema may want a
282
+ * number, a boolean or an object. Strings are tried first, so a tool that
283
+ * genuinely wants the string `"42"` still gets it; only if the schema refuses
284
+ * are JSON-looking values parsed.
285
+ *
286
+ * @param raw - Parameter name → raw text value
287
+ * @param schema - The tool's input schema, when it has one
288
+ * @returns The arguments, and the schema's complaint if it still refuses them
289
+ *
290
+ * @example
291
+ * coerceArgs({ timeout: '5000' }, z.object({ timeout: z.number() }))
292
+ * // → { args: { timeout: 5000 } }
293
+ */
294
+ export function coerceArgs(raw: Record<string, string>, schema?: SchemaLike): CoercedArgs {
295
+ const coerced: Record<string, unknown> = {};
296
+ for (const [key, value] of Object.entries(raw)) {
297
+ const trimmed = value.trim();
298
+ if (JSONISH.test(trimmed)) {
299
+ try {
300
+ coerced[key] = JSON.parse(trimmed);
301
+ continue;
302
+ } catch {
303
+ // Not valid JSON after all — keep the text.
304
+ }
305
+ }
306
+ coerced[key] = value;
307
+ }
308
+
309
+ if (!schema || typeof schema.safeParse !== 'function') return { args: coerced };
310
+ if (schema.safeParse(raw).success) return { args: raw };
311
+
312
+ const second = schema.safeParse(coerced);
313
+ if (second.success) return { args: coerced };
314
+
315
+ return { args: coerced, error: second.error?.message ?? 'arguments did not match the tool schema' };
316
+ }
@@ -0,0 +1,190 @@
1
+ /**
2
+ * Tests for the runtime half of text-emitted tool-call recovery: executing
3
+ * what the model wrote as prose and feeding the results back into the turn.
4
+ *
5
+ * The bug these lock down: deepseek-chat wrote `<||DSML|| invoke …>` into
6
+ * its reply, the SDK saw no tool call, and the step ended having done
7
+ * nothing — the agent said it would create a team and created nothing
8
+ * (2026-09-19).
9
+ */
10
+
11
+ import { describe, it, expect, vi, beforeEach } from 'vitest';
12
+ import { AgentRunnerService } from './agent-runner.service.js';
13
+ import { CREWLY_AGENT_DEFAULTS, type AgentRunResult, type CrewlyAgentConfig } from './types.js';
14
+
15
+ const MAX_STEPS = 50;
16
+
17
+ /** A run result with sensible defaults. */
18
+ function runResult(over: Partial<AgentRunResult> = {}): AgentRunResult {
19
+ return { text: 'text', steps: 1, usage: { input: 10, output: 5 }, toolCalls: [], finishReason: 'stop', ...over };
20
+ }
21
+
22
+ /** The envelope deepseek-chat actually emits, around one call. */
23
+ function dsml(tool: string, param: string, value: string): string {
24
+ return [
25
+ '<||DSML|| calls>',
26
+ `<||DSML|| invoke name="${tool}">`,
27
+ `<||DSML|| parameter name="${param}" string="true">${value}</||DSML|| parameter>`,
28
+ '</||DSML|| invoke>',
29
+ '</||DSML|| calls>',
30
+ ].join('\n');
31
+ }
32
+
33
+ describe('salvaging tool calls the model wrote as text', () => {
34
+ let runner: AgentRunnerService;
35
+ let attempts: AgentRunResult[];
36
+ let attemptSpy: ReturnType<typeof vi.fn>;
37
+ let bashExec: ReturnType<typeof vi.fn>;
38
+ let tools: Record<string, unknown>;
39
+
40
+ const config = {
41
+ sessionName: 'test-agent',
42
+ role: 'developer',
43
+ projectPath: '/tmp',
44
+ maxSteps: MAX_STEPS,
45
+ model: { provider: 'deepseek', modelId: 'deepseek-chat' },
46
+ } as unknown as CrewlyAgentConfig;
47
+
48
+ /** Drive the private loop with a scripted sequence of attempts. */
49
+ async function runLoop(): Promise<AgentRunResult> {
50
+ return await (runner as unknown as {
51
+ executeRunWithStreamText(tools: Record<string, unknown>, signal: AbortSignal): Promise<AgentRunResult>;
52
+ }).executeRunWithStreamText(tools, new AbortController().signal);
53
+ }
54
+
55
+ /** The messages the runner has accumulated for the turn. */
56
+ function messages(): Array<{ role: string; content: string }> {
57
+ return (runner as unknown as { state: { messages: Array<{ role: string; content: string }> } }).state.messages;
58
+ }
59
+
60
+ beforeEach(() => {
61
+ runner = new AgentRunnerService(config);
62
+ attempts = [];
63
+ attemptSpy = vi.fn(async () => {
64
+ const next = attempts.shift() ?? runResult();
65
+ // Mirror what a real attempt does: its text lands in the transcript.
66
+ if (next.text) messages().push({ role: 'assistant', content: next.text });
67
+ return next;
68
+ });
69
+ (runner as unknown as Record<string, unknown>).attemptWithErrorRetries = attemptSpy;
70
+ bashExec = vi.fn(async () => ({ success: true, stdout: 'M src/index.ts', exitCode: 0 }));
71
+ tools = { bash_exec: { description: 'run a command', inputSchema: undefined, execute: bashExec } };
72
+ vi.spyOn(console, 'warn').mockImplementation(() => undefined);
73
+ });
74
+
75
+ it('executes the call, hides the markup, and lets the turn finish', async () => {
76
+ attempts = [
77
+ runResult({ text: `Checking the repo.\n${dsml('bash_exec', 'command', 'git status --short')}` }),
78
+ runResult({ text: 'One file is modified.' }),
79
+ ];
80
+ const out = await runLoop();
81
+
82
+ expect(bashExec).toHaveBeenCalledWith({ command: 'git status --short' });
83
+ expect(out.text).toBe('Checking the repo.\n\nOne file is modified.');
84
+ expect(out.text).not.toContain('DSML');
85
+ expect(out.incomplete).toBeUndefined();
86
+ });
87
+
88
+ it('records the salvaged call in the run ledger, so it is not invisible work', async () => {
89
+ attempts = [runResult({ text: dsml('bash_exec', 'command', 'ls') }), runResult({ text: 'done' })];
90
+ const out = await runLoop();
91
+ expect(out.toolCalls).toEqual([
92
+ { toolName: 'bash_exec', args: { command: 'ls' }, result: { success: true, stdout: 'M src/index.ts', exitCode: 0 } },
93
+ ]);
94
+ });
95
+
96
+ it('feeds the result back and tells the model to use real tool calls', async () => {
97
+ attempts = [runResult({ text: dsml('bash_exec', 'command', 'ls') }), runResult({ text: 'done' })];
98
+ await runLoop();
99
+ const fedBack = messages().filter((m) => m.role === 'user').pop();
100
+ expect(fedBack?.content).toContain('M src/index.ts');
101
+ expect(fedBack?.content).toMatch(/executes nothing/i);
102
+ });
103
+
104
+ it('rewrites the transcript so the markup is not modelled as the way to call a tool', async () => {
105
+ attempts = [runResult({ text: `Checking.\n${dsml('bash_exec', 'command', 'ls')}` }), runResult({ text: 'done' })];
106
+ await runLoop();
107
+ expect(messages().some((m) => m.content.includes('DSML'))).toBe(false);
108
+ expect(messages().some((m) => m.role === 'assistant' && m.content === 'Checking.')).toBe(true);
109
+ });
110
+
111
+ it("runs the tool under the name another harness uses — the live failure was 'Bash'", async () => {
112
+ attempts = [runResult({ text: dsml('Bash', 'command', 'ls') }), runResult({ text: 'ok' })];
113
+ await runLoop();
114
+ expect(bashExec).toHaveBeenCalledWith({ command: 'ls' });
115
+ const fedBack = messages().filter((m) => m.role === 'user').pop();
116
+ expect(fedBack?.content).toContain('bash_exec (you wrote "Bash")');
117
+ });
118
+
119
+ it('reports a name it cannot resolve instead of guessing at a tool', async () => {
120
+ attempts = [runResult({ text: dsml('deploy_to_prod', 'target', 'live') }), runResult({ text: 'ok' })];
121
+ await runLoop();
122
+ expect(bashExec).not.toHaveBeenCalled();
123
+ const fedBack = messages().filter((m) => m.role === 'user').pop();
124
+ expect(fedBack?.content).toMatch(/no tool with that name/i);
125
+ expect(fedBack?.content).toContain('bash_exec');
126
+ });
127
+
128
+ it('reports a schema complaint rather than calling the tool with bad arguments', async () => {
129
+ tools = {
130
+ bash_exec: {
131
+ execute: bashExec,
132
+ inputSchema: { safeParse: () => ({ success: false, error: { message: 'command is required' } }) },
133
+ },
134
+ };
135
+ attempts = [runResult({ text: dsml('bash_exec', 'cmd', 'ls') }), runResult({ text: 'ok' })];
136
+ await runLoop();
137
+ expect(bashExec).not.toHaveBeenCalled();
138
+ expect(messages().filter((m) => m.role === 'user').pop()?.content).toContain('command is required');
139
+ });
140
+
141
+ it('hands a thrown tool error back as a result instead of losing the turn', async () => {
142
+ bashExec.mockRejectedValueOnce(new Error('spawn ENOENT'));
143
+ attempts = [runResult({ text: dsml('bash_exec', 'command', 'nope') }), runResult({ text: 'ok' })];
144
+ const out = await runLoop();
145
+ expect(out.toolCalls[0].result).toEqual({ error: 'spawn ENOENT' });
146
+ expect(messages().filter((m) => m.role === 'user').pop()?.content).toContain('spawn ENOENT');
147
+ });
148
+
149
+ it('gives up after the salvage budget rather than looping on a model that will not learn', async () => {
150
+ attempts = Array.from({ length: 12 }, () => runResult({ text: dsml('bash_exec', 'command', 'ls') }));
151
+ await runLoop();
152
+ expect(bashExec).toHaveBeenCalledTimes(CREWLY_AGENT_DEFAULTS.MAX_TEXT_TOOL_SALVAGES);
153
+ });
154
+
155
+ it('caps how many calls one confused turn can fan out to', async () => {
156
+ const many = Array.from({ length: 9 }, (_, i) => `<invoke name="bash_exec"><parameter name="command">echo ${i}</parameter></invoke>`).join('\n');
157
+ attempts = [runResult({ text: many }), runResult({ text: 'ok' })];
158
+ await runLoop();
159
+ expect(bashExec).toHaveBeenCalledTimes(CREWLY_AGENT_DEFAULTS.MAX_SALVAGED_CALLS_PER_ROUND);
160
+ expect(messages().filter((m) => m.role === 'user').pop()?.content).toMatch(/4 further call\(s\) were not run/);
161
+ });
162
+
163
+ it('truncates a huge result so one salvaged command cannot blow the context', async () => {
164
+ bashExec.mockResolvedValueOnce('x'.repeat(CREWLY_AGENT_DEFAULTS.SALVAGED_RESULT_MAX_CHARS + 500));
165
+ attempts = [runResult({ text: dsml('bash_exec', 'command', 'cat big') }), runResult({ text: 'ok' })];
166
+ await runLoop();
167
+ const fedBack = messages().filter((m) => m.role === 'user').pop()?.content ?? '';
168
+ expect(fedBack).toMatch(/truncated, 500 more characters/);
169
+ expect(fedBack.length).toBeLessThan(CREWLY_AGENT_DEFAULTS.SALVAGED_RESULT_MAX_CHARS + 1500);
170
+ });
171
+
172
+ it('leaves a healthy turn completely alone', async () => {
173
+ attempts = [runResult({ text: 'All done, nothing to salvage.' })];
174
+ const out = await runLoop();
175
+ expect(attemptSpy).toHaveBeenCalledTimes(1);
176
+ expect(bashExec).not.toHaveBeenCalled();
177
+ expect(out.text).toBe('All done, nothing to salvage.');
178
+ });
179
+
180
+ it('still reports the turn incomplete when the retried attempt bails', async () => {
181
+ attempts = [
182
+ runResult({ text: dsml('bash_exec', 'command', 'ls') }),
183
+ runResult({ text: '', finishReason: 'other' }),
184
+ runResult({ text: '', finishReason: 'other' }),
185
+ ];
186
+ const out = await runLoop();
187
+ expect(bashExec).toHaveBeenCalledTimes(1);
188
+ expect(out.incomplete).toMatchObject({ reason: 'abnormal-finish' });
189
+ });
190
+ });