crewly 1.20.35 → 1.20.40

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. package/config/skills/_common/lib.sh +6 -0
  2. package/config/skills/agent/core/calendar-create/SKILL.md +10 -0
  3. package/config/skills/agent/core/calendar-create/execute.sh +6 -0
  4. package/config/skills/agent/core/calendar-list/SKILL.md +10 -0
  5. package/config/skills/agent/core/calendar-list/execute.sh +6 -0
  6. package/config/skills/agent/core/docs-read/SKILL.md +10 -0
  7. package/config/skills/agent/core/docs-read/execute.sh +6 -1
  8. package/config/skills/agent/core/docs-write/SKILL.md +10 -0
  9. package/config/skills/agent/core/docs-write/execute.sh +6 -1
  10. package/config/skills/agent/core/drive-read/SKILL.md +10 -0
  11. package/config/skills/agent/core/drive-read/execute.sh +6 -1
  12. package/config/skills/agent/core/drive-search/SKILL.md +10 -0
  13. package/config/skills/agent/core/drive-search/execute.sh +6 -1
  14. package/config/skills/agent/core/drive-upload/SKILL.md +10 -0
  15. package/config/skills/agent/core/drive-upload/execute.sh +6 -1
  16. package/config/skills/agent/core/gmail-read/SKILL.md +10 -0
  17. package/config/skills/agent/core/gmail-read/execute.sh +6 -0
  18. package/config/skills/agent/core/gmail-search/SKILL.md +10 -0
  19. package/config/skills/agent/core/gmail-search/execute.sh +6 -0
  20. package/config/skills/agent/core/gmail-send/SKILL.md +10 -0
  21. package/config/skills/agent/core/gmail-send/execute.sh +6 -0
  22. package/config/skills/agent/core/sheets-read/SKILL.md +10 -0
  23. package/config/skills/agent/core/sheets-read/execute.sh +6 -1
  24. package/config/skills/agent/core/sheets-write/SKILL.md +10 -0
  25. package/config/skills/agent/core/sheets-write/execute.sh +6 -1
  26. package/config/skills/agent/core/slides-create/SKILL.md +10 -0
  27. package/config/skills/agent/core/slides-create/execute.sh +6 -1
  28. package/config/skills/agent/core/slides-read/SKILL.md +10 -0
  29. package/config/skills/agent/core/slides-read/execute.sh +6 -1
  30. package/config/slack-app-manifest.json +16 -9
  31. package/dist/backend/backend/src/constants.d.ts +18 -4
  32. package/dist/backend/backend/src/constants.d.ts.map +1 -1
  33. package/dist/backend/backend/src/constants.js +16 -4
  34. package/dist/backend/backend/src/constants.js.map +1 -1
  35. package/dist/backend/backend/src/controllers/google/google.controller.d.ts +8 -0
  36. package/dist/backend/backend/src/controllers/google/google.controller.d.ts.map +1 -1
  37. package/dist/backend/backend/src/controllers/google/google.controller.js +137 -37
  38. package/dist/backend/backend/src/controllers/google/google.controller.js.map +1 -1
  39. package/dist/backend/backend/src/controllers/google/google.routes.d.ts +2 -1
  40. package/dist/backend/backend/src/controllers/google/google.routes.d.ts.map +1 -1
  41. package/dist/backend/backend/src/controllers/google/google.routes.js +4 -2
  42. package/dist/backend/backend/src/controllers/google/google.routes.js.map +1 -1
  43. package/dist/backend/backend/src/controllers/slack/slack-error.utils.d.ts +46 -0
  44. package/dist/backend/backend/src/controllers/slack/slack-error.utils.d.ts.map +1 -0
  45. package/dist/backend/backend/src/controllers/slack/slack-error.utils.js +54 -0
  46. package/dist/backend/backend/src/controllers/slack/slack-error.utils.js.map +1 -0
  47. package/dist/backend/backend/src/controllers/slack/slack.controller.d.ts.map +1 -1
  48. package/dist/backend/backend/src/controllers/slack/slack.controller.js +5 -12
  49. package/dist/backend/backend/src/controllers/slack/slack.controller.js.map +1 -1
  50. package/dist/backend/backend/src/services/google/google-api.client.d.ts +23 -2
  51. package/dist/backend/backend/src/services/google/google-api.client.d.ts.map +1 -1
  52. package/dist/backend/backend/src/services/google/google-api.client.js +5 -2
  53. package/dist/backend/backend/src/services/google/google-api.client.js.map +1 -1
  54. package/dist/backend/backend/src/services/google/google-workspace-token.service.d.ts +61 -11
  55. package/dist/backend/backend/src/services/google/google-workspace-token.service.d.ts.map +1 -1
  56. package/dist/backend/backend/src/services/google/google-workspace-token.service.js +108 -31
  57. package/dist/backend/backend/src/services/google/google-workspace-token.service.js.map +1 -1
  58. package/dist/backend/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
  59. package/dist/backend/backend/src/services/slack/slack-team-channel.service.js +27 -4
  60. package/dist/backend/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
  61. package/dist/backend/build-info.json +2 -2
  62. package/dist/cli/backend/src/constants.d.ts +18 -4
  63. package/dist/cli/backend/src/constants.d.ts.map +1 -1
  64. package/dist/cli/backend/src/constants.js +16 -4
  65. package/dist/cli/backend/src/constants.js.map +1 -1
  66. package/dist/cli/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
  67. package/dist/cli/backend/src/services/slack/slack-team-channel.service.js +27 -4
  68. package/dist/cli/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
  69. package/frontend/dist/assets/{index-e079a375.js → index-e7785269.js} +267 -267
  70. package/frontend/dist/index.html +1 -1
  71. package/package.json +1 -1
  72. package/packages/crewly-agent/src/runtime/agent-runner.service.test.ts +10 -1
  73. package/packages/crewly-agent/src/runtime/agent-runner.service.ts +179 -3
  74. package/packages/crewly-agent/src/runtime/text-tool-calls.test.ts +144 -0
  75. package/packages/crewly-agent/src/runtime/text-tool-calls.ts +316 -0
  76. package/packages/crewly-agent/src/runtime/text-tool-salvage.test.ts +190 -0
  77. package/packages/crewly-agent/src/runtime/types.ts +6 -0
@@ -7,7 +7,7 @@
7
7
  <meta name="color-scheme" content="dark" />
8
8
  <!-- Nunito font is self-hosted via @fontsource/nunito (imported in main.tsx) -->
9
9
  <title>Crewly AI Studio</title>
10
- <script type="module" crossorigin src="/assets/index-e079a375.js"></script>
10
+ <script type="module" crossorigin src="/assets/index-e7785269.js"></script>
11
11
  <link rel="stylesheet" href="/assets/index-159eab4f.css">
12
12
  </head>
13
13
  <body class="bg-background-dark font-display text-text-primary-dark">
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "crewly",
3
- "version": "1.20.35",
3
+ "version": "1.20.40",
4
4
  "type": "module",
5
5
  "description": "Multi-agent orchestration platform for AI coding teams — coordinates Claude Code, Gemini CLI, and Codex agents with a real-time web dashboard",
6
6
  "workspaces": [
@@ -55,9 +55,18 @@ describe('AgentRunnerService', () => {
55
55
  it('should initialize conversation state with empty messages', () => {
56
56
  const state = runner.getState();
57
57
  expect(state.messages).toEqual([]);
58
- expect(state.systemPrompt).toBe('You are a test agent.');
58
+ expect(state.systemPrompt).toContain('You are a test agent.');
59
59
  expect(state.totalTokens).toEqual({ input: 0, output: 0 });
60
60
  });
61
+
62
+ it('appends the harness rules to every role prompt, so a weak model is told how to work', () => {
63
+ const prompt = runner.getState().systemPrompt;
64
+ expect(prompt.indexOf('You are a test agent.')).toBeLessThan(prompt.indexOf('## How to work'));
65
+ expect(prompt).toMatch(/never write a tool invocation as text/i);
66
+ expect(prompt).toMatch(/do the work in this turn/i);
67
+ // Naming the markup is what teaches a model to emit it.
68
+ expect(prompt).not.toMatch(/<\s*\/?\s*(invoke|parameter|function_calls)/i);
69
+ });
61
70
  });
62
71
 
63
72
  describe('initialize', () => {
@@ -15,6 +15,7 @@ import { createTools } from './tool-registry.js';
15
15
  import { connectAndLoadMcpTools } from './mcp-tool-bridge.js';
16
16
  import { ApprovalQueueService, type PendingApproval } from './approval-queue.service.js';
17
17
  import { OutputFilterService } from './output-filter.service.js';
18
+ import { parseTextToolCalls, coerceArgs, resolveToolName, type TextToolCall, type SchemaLike } from './text-tool-calls.js';
18
19
  import type { ToolDefinition, McpClientLike } from './types.js';
19
20
  import {
20
21
  type CrewlyAgentConfig,
@@ -482,6 +483,25 @@ export class AgentRunnerService {
482
483
  * @param modelManager - Optional model manager instance (for testing)
483
484
  * @param apiClient - Optional API client instance (for testing)
484
485
  */
486
+ /**
487
+ * Rules about *how* to work that every role prompt gets, whatever the model.
488
+ *
489
+ * These exist because a weaker model fails in ways a strong one does not:
490
+ * it writes its tool call as prose (so nothing runs and the user sees
491
+ * markup), or it narrates a plan and stops without carrying it out. The
492
+ * runtime recovers from both, but saying so plainly costs a few tokens and
493
+ * prevents most of it. The wording deliberately never shows the markup
494
+ * syntax — describing it is what teaches a model to emit it.
495
+ */
496
+ private static readonly HARNESS_RULES = [
497
+ '## How to work',
498
+ '',
499
+ '- Call tools through the tool-calling mechanism. Never write a tool invocation as text: your text is shown to the user verbatim and executes nothing.',
500
+ '- Do the work in this turn. If you say you will do something, do it before you finish — a plan with no action is a failed turn.',
501
+ '- Act, then check. Run the tool, read the result, and continue from what it actually returned rather than from what you expected.',
502
+ '- If something blocks you, say what blocked you and what you tried. Never report work as done that you did not verify.',
503
+ ].join('\n');
504
+
485
505
  constructor(
486
506
  config: CrewlyAgentConfig,
487
507
  modelManager?: ModelManager,
@@ -495,9 +515,10 @@ export class AgentRunnerService {
495
515
  );
496
516
  this.securityPolicy = { ...CREWLY_AGENT_DEFAULTS.SECURITY_POLICY };
497
517
  // In eval mode, strip delegation-first instructions so agent implements directly
498
- this.effectiveSystemPrompt = config.evalMode
518
+ const rolePrompt = config.evalMode
499
519
  ? AgentRunnerService.stripDelegationInstructions(config.systemPrompt)
500
520
  : config.systemPrompt;
521
+ this.effectiveSystemPrompt = `${rolePrompt}\n\n${AgentRunnerService.HARNESS_RULES}`;
501
522
  // Conversation states are lazy-created on first access via the
502
523
  // `state` getter, so we don't need to seed `__default__` here.
503
524
  // The first message processed will create whichever conversation
@@ -1188,7 +1209,7 @@ export class AgentRunnerService {
1188
1209
  tools: Record<string, unknown>,
1189
1210
  abortSignal: AbortSignal,
1190
1211
  ): Promise<AgentRunResult> {
1191
- let result = await this.attemptWithErrorRetries(tools, abortSignal);
1212
+ let result = await this.attemptWithSalvage(tools, abortSignal);
1192
1213
  let outcome = classifyFinish(result.finishReason, result.steps, this.config.maxSteps);
1193
1214
  let recoveryAttempts = 0;
1194
1215
 
@@ -1201,7 +1222,7 @@ export class AgentRunnerService {
1201
1222
  recoveryAttempts++;
1202
1223
  this.streamingCallbacks.onTextChunk?.(`[recover] ${outcome.reason} — continuing (${recoveryAttempts}/${outcome.budget})\n`);
1203
1224
  this.state.messages.push({ role: 'user', content: outcome.nudge });
1204
- const next = await this.attemptWithErrorRetries(tools, abortSignal);
1225
+ const next = await this.attemptWithSalvage(tools, abortSignal);
1205
1226
  result = mergeRuns(result, next);
1206
1227
  outcome = classifyFinish(next.finishReason, next.steps, this.config.maxSteps);
1207
1228
  }
@@ -1219,6 +1240,161 @@ export class AgentRunnerService {
1219
1240
  };
1220
1241
  }
1221
1242
 
1243
+ /**
1244
+ * Run one turn, then rescue any tool call the model *wrote* instead of called.
1245
+ *
1246
+ * A weak model sometimes emits its call envelope into the text channel:
1247
+ * the provider returns prose, the SDK sees no tool call, and the step ends
1248
+ * having done nothing — the failure mode behind both the markup users saw
1249
+ * in Slack and the turns that promised work and produced none. Rather than
1250
+ * strip the markup and lose the intent, the envelope is parsed, the tools
1251
+ * are executed for real, and the results are handed back so the turn can
1252
+ * carry on.
1253
+ *
1254
+ * Salvaged calls go through the tool's own `execute`, so approval gates and
1255
+ * command blocklists apply exactly as they do to a native call — this
1256
+ * recovers lost work, it does not widen what the agent may do.
1257
+ *
1258
+ * @param tools - The tool registry for this run
1259
+ * @param abortSignal - Cancels the turn and any further rounds
1260
+ * @returns The turn's result, with salvaged calls folded into `toolCalls`
1261
+ */
1262
+ private async attemptWithSalvage(
1263
+ tools: Record<string, unknown>,
1264
+ abortSignal: AbortSignal,
1265
+ ): Promise<AgentRunResult> {
1266
+ let attempt = await this.attemptWithErrorRetries(tools, abortSignal);
1267
+ const salvaged: ToolCallRecord[] = [];
1268
+
1269
+ for (let round = 0; round < CREWLY_AGENT_DEFAULTS.MAX_TEXT_TOOL_SALVAGES; round++) {
1270
+ if (abortSignal.aborted) break;
1271
+ const parsed = parseTextToolCalls(attempt.text ?? '');
1272
+ if (parsed.calls.length === 0) break;
1273
+
1274
+ console.warn('[AgentRunner] Model wrote tool calls as text — executing them:', {
1275
+ round: round + 1,
1276
+ tools: parsed.calls.map(c => c.toolName),
1277
+ });
1278
+ this.streamingCallbacks.onTextChunk?.(
1279
+ `[salvage] running ${parsed.calls.length} tool call(s) the model wrote as text\n`,
1280
+ );
1281
+
1282
+ const { records, report } = await this.runSalvagedCalls(parsed.calls, tools);
1283
+ salvaged.push(...records);
1284
+
1285
+ // Rewrite the turn's own message so the transcript does not keep
1286
+ // teaching the model that writing markup is how a tool gets called.
1287
+ this.replaceLastAssistantMessage(parsed.text || '(I wrote a tool call as text instead of calling the tool.)');
1288
+ this.state.messages.push({ role: 'user', content: report });
1289
+
1290
+ const next = await this.attemptWithErrorRetries(tools, abortSignal);
1291
+ attempt = mergeRuns({ ...attempt, text: parsed.text }, next);
1292
+ }
1293
+
1294
+ return salvaged.length > 0
1295
+ ? { ...attempt, toolCalls: [...attempt.toolCalls, ...salvaged] }
1296
+ : attempt;
1297
+ }
1298
+
1299
+ /**
1300
+ * Execute the tool calls recovered from text and describe the results.
1301
+ *
1302
+ * Unknown tools and arguments the schema rejects are reported back rather
1303
+ * than guessed at: the model gets a specific complaint it can act on, which
1304
+ * is far more useful than silence.
1305
+ *
1306
+ * @param calls - Calls parsed out of the model's text
1307
+ * @param tools - The tool registry for this run
1308
+ * @returns Records for the run ledger, and the message to feed back
1309
+ */
1310
+ private async runSalvagedCalls(
1311
+ calls: TextToolCall[],
1312
+ tools: Record<string, unknown>,
1313
+ ): Promise<{ records: ToolCallRecord[]; report: string }> {
1314
+ const records: ToolCallRecord[] = [];
1315
+ const sections: string[] = [];
1316
+ const registry = tools as Record<string, ToolDefinition | undefined>;
1317
+
1318
+ for (const call of calls.slice(0, CREWLY_AGENT_DEFAULTS.MAX_SALVAGED_CALLS_PER_ROUND)) {
1319
+ // A model that has seen another harness asks for `Bash`, not `bash_exec`.
1320
+ const resolved = resolveToolName(call.toolName, Object.keys(registry));
1321
+ const def = resolved ? registry[resolved] : undefined;
1322
+ if (!resolved || !def || typeof def.execute !== 'function') {
1323
+ sections.push(`### ${call.toolName}\nThere is no tool with that name. Available tools: ${Object.keys(registry).join(', ')}`);
1324
+ continue;
1325
+ }
1326
+
1327
+ const { args, error } = coerceArgs(call.args, def.inputSchema as unknown as SchemaLike | undefined);
1328
+ if (error) {
1329
+ sections.push(`### ${resolved}\nThe arguments were rejected: ${error}`);
1330
+ continue;
1331
+ }
1332
+
1333
+ const startedAt = Date.now();
1334
+ this.streamingCallbacks.onToolCallStart?.(resolved, args);
1335
+ let output: unknown;
1336
+ try {
1337
+ output = await def.execute(args);
1338
+ } catch (err) {
1339
+ output = { error: err instanceof Error ? err.message : String(err) };
1340
+ }
1341
+ this.streamingCallbacks.onToolCallFinish?.(resolved, args, output, Date.now() - startedAt);
1342
+
1343
+ records.push({ toolName: resolved, args, result: output });
1344
+ // Name it as the model wrote it when that differed, so it learns the real name.
1345
+ const heading = resolved === call.toolName ? resolved : `${resolved} (you wrote "${call.toolName}")`;
1346
+ sections.push(`### ${heading}\n${this.summarizeSalvagedResult(output)}`);
1347
+ }
1348
+
1349
+ const skipped = calls.length - Math.min(calls.length, CREWLY_AGENT_DEFAULTS.MAX_SALVAGED_CALLS_PER_ROUND);
1350
+ const report = [
1351
+ 'You wrote your tool calls as text, so the model API never received them. I executed them for you; here is what they returned.',
1352
+ ...sections,
1353
+ skipped > 0 ? `(${skipped} further call(s) were not run — make them yourself.)` : '',
1354
+ 'Use the tool-calling mechanism from now on: text in your reply is shown to the user verbatim and executes nothing. Continue the task with these results.',
1355
+ ].filter(Boolean).join('\n\n');
1356
+
1357
+ return { records, report };
1358
+ }
1359
+
1360
+ /**
1361
+ * Render a salvaged tool result small enough to feed back.
1362
+ *
1363
+ * @param output - Whatever the tool returned
1364
+ * @returns A string, truncated with a note when it was long
1365
+ */
1366
+ private summarizeSalvagedResult(output: unknown): string {
1367
+ let rendered: string;
1368
+ try {
1369
+ rendered = typeof output === 'string' ? output : JSON.stringify(output, null, 2) ?? String(output);
1370
+ } catch {
1371
+ rendered = String(output);
1372
+ }
1373
+ const limit = CREWLY_AGENT_DEFAULTS.SALVAGED_RESULT_MAX_CHARS;
1374
+ return rendered.length > limit
1375
+ ? `${rendered.slice(0, limit)}\n… (truncated, ${rendered.length - limit} more characters)`
1376
+ : rendered;
1377
+ }
1378
+
1379
+ /**
1380
+ * Rewrite the assistant message this turn just added to the transcript.
1381
+ *
1382
+ * Stops at the user message that opened the turn, so an earlier, healthy
1383
+ * reply is never touched.
1384
+ *
1385
+ * @param content - Replacement content
1386
+ */
1387
+ private replaceLastAssistantMessage(content: string): void {
1388
+ for (let i = this.state.messages.length - 1; i >= 0; i--) {
1389
+ const message = this.state.messages[i];
1390
+ if (message.role === 'assistant') {
1391
+ this.state.messages[i] = { ...message, content };
1392
+ return;
1393
+ }
1394
+ if (message.role === 'user') return;
1395
+ }
1396
+ }
1397
+
1222
1398
  /**
1223
1399
  * Run one turn, retrying only on *thrown* failures (rate limits, network,
1224
1400
  * context length). A turn that returns with a bad `finishReason` is the
@@ -0,0 +1,144 @@
1
+ /**
2
+ * Tests for recovering tool calls a model wrote as text.
3
+ *
4
+ * The bug these lock down: deepseek-chat writes its call envelope into the
5
+ * text channel with its own separators (`<||DSML|| invoke name="Bash">`),
6
+ * so no tool ever ran and the user was shown the markup instead of an answer
7
+ * (2026-09-19). Matching is by shape — a new prefix must not defeat it.
8
+ */
9
+
10
+ import { describe, it, expect } from 'vitest';
11
+ import { parseTextToolCalls, hasTextToolCalls, coerceArgs, resolveToolName, type SchemaLike } from './text-tool-calls.js';
12
+
13
+ /** The exact bytes deepseek-chat emitted, taken from ~/.crewly/chat.db. */
14
+ const DSML = [
15
+ 'Let me check the repo.',
16
+ '<||DSML|| calls>',
17
+ '<||DSML|| invoke name="bash_exec">',
18
+ '<||DSML|| parameter name="command" string="true">git status --short</||DSML|| parameter>',
19
+ '<||DSML|| parameter name="timeout" string="true">5000</||DSML|| parameter>',
20
+ '</||DSML|| invoke>',
21
+ '</||DSML|| calls>',
22
+ ].join('\n');
23
+
24
+ /** A minimal Zod-style schema. */
25
+ function schema(check: (v: Record<string, unknown>) => boolean): SchemaLike {
26
+ return { safeParse: (v: unknown) => (check(v as Record<string, unknown>) ? { success: true, data: v } : { success: false, error: { message: 'bad args' } }) };
27
+ }
28
+
29
+ describe('parseTextToolCalls', () => {
30
+ it('recovers the deepseek envelope, prefix and all, and leaves the prose', () => {
31
+ const { calls, text } = parseTextToolCalls(DSML);
32
+ expect(calls).toEqual([
33
+ { toolName: 'bash_exec', args: { command: 'git status --short', timeout: '5000' } },
34
+ ]);
35
+ expect(text).toBe('Let me check the repo.');
36
+ });
37
+
38
+ it('recovers the plain Claude-style envelope too', () => {
39
+ const { calls } = parseTextToolCalls('<function_calls><invoke name="read_file"><parameter name="path">a.ts</parameter></invoke></function_calls>');
40
+ expect(calls).toEqual([{ toolName: 'read_file', args: { path: 'a.ts' } }]);
41
+ });
42
+
43
+ it('recovers every call in a batch, in order', () => {
44
+ const { calls } = parseTextToolCalls(
45
+ '<invoke name="a"><parameter name="x">1</parameter></invoke><invoke name="b"><parameter name="y">2</parameter></invoke>',
46
+ );
47
+ expect(calls.map((c) => c.toolName)).toEqual(['a', 'b']);
48
+ expect(calls[1].args).toEqual({ y: '2' });
49
+ });
50
+
51
+ it('keeps a multi-line value intact, including its own angle brackets', () => {
52
+ const { calls } = parseTextToolCalls(
53
+ '<invoke name="write_file"><parameter name="content">line 1\nif (a < b) { go(); }\nline 3</parameter></invoke>',
54
+ );
55
+ expect(calls[0].args.content).toBe('line 1\nif (a < b) { go(); }\nline 3');
56
+ });
57
+
58
+ it('salvages a call the model never closed', () => {
59
+ const { calls, text } = parseTextToolCalls('Working.\n<invoke name="bash_exec"><parameter name="command">npm test');
60
+ expect(calls).toEqual([{ toolName: 'bash_exec', args: { command: 'npm test' } }]);
61
+ expect(text).toBe('Working.');
62
+ });
63
+
64
+ it('leaves a fenced example alone — it is documentation, not a call', () => {
65
+ const raw = 'Here is the syntax:\n\n```\n<invoke name="bash_exec"><parameter name="command">ls</parameter></invoke>\n```';
66
+ const { calls, text } = parseTextToolCalls(raw);
67
+ expect(calls).toEqual([]);
68
+ expect(text).toBe(raw);
69
+ });
70
+
71
+ it('finds nothing in ordinary prose, and does not touch it', () => {
72
+ for (const prose of ['', 'The build recalls the cached layer.', 'Use <Parameter> in the docs? no.']) {
73
+ expect(parseTextToolCalls(prose).calls).toEqual([]);
74
+ }
75
+ expect(parseTextToolCalls('The build recalls the cached layer.').text).toBe('The build recalls the cached layer.');
76
+ });
77
+
78
+ it('ignores an invoke tag with no name rather than inventing one', () => {
79
+ expect(parseTextToolCalls('<invoke><parameter name="x">1</parameter></invoke>').calls).toEqual([]);
80
+ });
81
+ });
82
+
83
+ describe('hasTextToolCalls', () => {
84
+ it('is true for any prefix and false for prose', () => {
85
+ expect(hasTextToolCalls(DSML)).toBe(true);
86
+ expect(hasTextToolCalls('<invoke name="x">')).toBe(true);
87
+ expect(hasTextToolCalls('nothing to see')).toBe(false);
88
+ });
89
+ });
90
+
91
+ describe('coerceArgs', () => {
92
+ it('passes strings through when the schema accepts them', () => {
93
+ const out = coerceArgs({ command: '5' }, schema((v) => typeof v.command === 'string'));
94
+ expect(out).toEqual({ args: { command: '5' } });
95
+ });
96
+
97
+ it('parses JSON-looking values only when the schema refuses the strings', () => {
98
+ const out = coerceArgs({ timeout: '5000', on: 'true', tags: '["a"]' }, schema((v) => typeof v.timeout === 'number'));
99
+ expect(out.args).toEqual({ timeout: 5000, on: true, tags: ['a'] });
100
+ expect(out.error).toBeUndefined();
101
+ });
102
+
103
+ it('coerces with no schema to give the tool its best shot', () => {
104
+ expect(coerceArgs({ n: '3', s: 'hello' }).args).toEqual({ n: 3, s: 'hello' });
105
+ });
106
+
107
+ it('reports the schema complaint instead of calling a tool with bad arguments', () => {
108
+ const out = coerceArgs({ command: 'ls' }, schema((v) => typeof v.path === 'string'));
109
+ expect(out.error).toBe('bad args');
110
+ });
111
+
112
+ it('keeps a value that only looks like JSON', () => {
113
+ expect(coerceArgs({ command: '{ not json' }).args).toEqual({ command: '{ not json' });
114
+ });
115
+ });
116
+
117
+ describe('resolveToolName', () => {
118
+ const available = ['bash_exec', 'read_file', 'write_file', 'edit_file', 'glob', 'grep', 'delegate_task', 'git_status'];
119
+
120
+ it('takes an exact name unchanged', () => {
121
+ expect(resolveToolName('bash_exec', available)).toBe('bash_exec');
122
+ });
123
+
124
+ it('matches across casing and separators, so GitStatus finds git_status', () => {
125
+ for (const written of ['GitStatus', 'git-status', 'GIT_STATUS', 'gitStatus']) {
126
+ expect(resolveToolName(written, available)).toBe('git_status');
127
+ }
128
+ });
129
+
130
+ it("resolves the names another harness uses — the live failure was 'Bash'", () => {
131
+ expect(resolveToolName('Bash', available)).toBe('bash_exec');
132
+ expect(resolveToolName('Read', available)).toBe('read_file');
133
+ expect(resolveToolName('Write', available)).toBe('write_file');
134
+ expect(resolveToolName('Edit', available)).toBe('edit_file');
135
+ expect(resolveToolName('Task', available)).toBe('delegate_task');
136
+ expect(resolveToolName('Search', available)).toBe('grep');
137
+ });
138
+
139
+ it('does not invent a tool the run does not have', () => {
140
+ expect(resolveToolName('Bash', ['read_file'])).toBeUndefined();
141
+ expect(resolveToolName('deploy_to_prod', available)).toBeUndefined();
142
+ expect(resolveToolName('', available)).toBeUndefined();
143
+ });
144
+ });