crewly 1.20.34 → 1.20.40
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/config/skills/_common/lib.sh +6 -0
- package/config/skills/agent/core/calendar-create/SKILL.md +10 -0
- package/config/skills/agent/core/calendar-create/execute.sh +6 -0
- package/config/skills/agent/core/calendar-list/SKILL.md +10 -0
- package/config/skills/agent/core/calendar-list/execute.sh +6 -0
- package/config/skills/agent/core/docs-read/SKILL.md +10 -0
- package/config/skills/agent/core/docs-read/execute.sh +6 -1
- package/config/skills/agent/core/docs-write/SKILL.md +10 -0
- package/config/skills/agent/core/docs-write/execute.sh +6 -1
- package/config/skills/agent/core/drive-read/SKILL.md +10 -0
- package/config/skills/agent/core/drive-read/execute.sh +6 -1
- package/config/skills/agent/core/drive-search/SKILL.md +10 -0
- package/config/skills/agent/core/drive-search/execute.sh +6 -1
- package/config/skills/agent/core/drive-upload/SKILL.md +10 -0
- package/config/skills/agent/core/drive-upload/execute.sh +6 -1
- package/config/skills/agent/core/gmail-read/SKILL.md +10 -0
- package/config/skills/agent/core/gmail-read/execute.sh +6 -0
- package/config/skills/agent/core/gmail-search/SKILL.md +10 -0
- package/config/skills/agent/core/gmail-search/execute.sh +6 -0
- package/config/skills/agent/core/gmail-send/SKILL.md +10 -0
- package/config/skills/agent/core/gmail-send/execute.sh +6 -0
- package/config/skills/agent/core/sheets-read/SKILL.md +10 -0
- package/config/skills/agent/core/sheets-read/execute.sh +6 -1
- package/config/skills/agent/core/sheets-write/SKILL.md +10 -0
- package/config/skills/agent/core/sheets-write/execute.sh +6 -1
- package/config/skills/agent/core/slides-create/SKILL.md +10 -0
- package/config/skills/agent/core/slides-create/execute.sh +6 -1
- package/config/skills/agent/core/slides-read/SKILL.md +10 -0
- package/config/skills/agent/core/slides-read/execute.sh +6 -1
- package/config/slack-app-manifest.json +16 -9
- package/dist/backend/backend/src/constants.d.ts +18 -4
- package/dist/backend/backend/src/constants.d.ts.map +1 -1
- package/dist/backend/backend/src/constants.js +16 -4
- package/dist/backend/backend/src/constants.js.map +1 -1
- package/dist/backend/backend/src/controllers/google/google.controller.d.ts +8 -0
- package/dist/backend/backend/src/controllers/google/google.controller.d.ts.map +1 -1
- package/dist/backend/backend/src/controllers/google/google.controller.js +137 -37
- package/dist/backend/backend/src/controllers/google/google.controller.js.map +1 -1
- package/dist/backend/backend/src/controllers/google/google.routes.d.ts +2 -1
- package/dist/backend/backend/src/controllers/google/google.routes.d.ts.map +1 -1
- package/dist/backend/backend/src/controllers/google/google.routes.js +4 -2
- package/dist/backend/backend/src/controllers/google/google.routes.js.map +1 -1
- package/dist/backend/backend/src/controllers/slack/slack-error.utils.d.ts +46 -0
- package/dist/backend/backend/src/controllers/slack/slack-error.utils.d.ts.map +1 -0
- package/dist/backend/backend/src/controllers/slack/slack-error.utils.js +54 -0
- package/dist/backend/backend/src/controllers/slack/slack-error.utils.js.map +1 -0
- package/dist/backend/backend/src/controllers/slack/slack.controller.d.ts.map +1 -1
- package/dist/backend/backend/src/controllers/slack/slack.controller.js +5 -12
- package/dist/backend/backend/src/controllers/slack/slack.controller.js.map +1 -1
- package/dist/backend/backend/src/services/google/google-api.client.d.ts +23 -2
- package/dist/backend/backend/src/services/google/google-api.client.d.ts.map +1 -1
- package/dist/backend/backend/src/services/google/google-api.client.js +5 -2
- package/dist/backend/backend/src/services/google/google-api.client.js.map +1 -1
- package/dist/backend/backend/src/services/google/google-workspace-token.service.d.ts +61 -11
- package/dist/backend/backend/src/services/google/google-workspace-token.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/google/google-workspace-token.service.js +108 -31
- package/dist/backend/backend/src/services/google/google-workspace-token.service.js.map +1 -1
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.js +27 -4
- package/dist/backend/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
- package/dist/backend/build-info.json +2 -2
- package/dist/cli/backend/src/constants.d.ts +18 -4
- package/dist/cli/backend/src/constants.d.ts.map +1 -1
- package/dist/cli/backend/src/constants.js +16 -4
- package/dist/cli/backend/src/constants.js.map +1 -1
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.d.ts.map +1 -1
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.js +27 -4
- package/dist/cli/backend/src/services/slack/slack-team-channel.service.js.map +1 -1
- package/frontend/dist/assets/{index-e079a375.js → index-e7785269.js} +267 -267
- package/frontend/dist/index.html +1 -1
- package/package.json +1 -1
- package/packages/crewly-agent/src/runtime/agent-runner.service.test.ts +10 -1
- package/packages/crewly-agent/src/runtime/agent-runner.service.ts +321 -1
- package/packages/crewly-agent/src/runtime/finish-recovery.test.ts +177 -0
- package/packages/crewly-agent/src/runtime/text-tool-calls.test.ts +144 -0
- package/packages/crewly-agent/src/runtime/text-tool-calls.ts +316 -0
- package/packages/crewly-agent/src/runtime/text-tool-salvage.test.ts +190 -0
- package/packages/crewly-agent/src/runtime/types.ts +38 -0
package/frontend/dist/index.html
CHANGED
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
<meta name="color-scheme" content="dark" />
|
|
8
8
|
<!-- Nunito font is self-hosted via @fontsource/nunito (imported in main.tsx) -->
|
|
9
9
|
<title>Crewly AI Studio</title>
|
|
10
|
-
<script type="module" crossorigin src="/assets/index-
|
|
10
|
+
<script type="module" crossorigin src="/assets/index-e7785269.js"></script>
|
|
11
11
|
<link rel="stylesheet" href="/assets/index-159eab4f.css">
|
|
12
12
|
</head>
|
|
13
13
|
<body class="bg-background-dark font-display text-text-primary-dark">
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "crewly",
|
|
3
|
-
"version": "1.20.
|
|
3
|
+
"version": "1.20.40",
|
|
4
4
|
"type": "module",
|
|
5
5
|
"description": "Multi-agent orchestration platform for AI coding teams — coordinates Claude Code, Gemini CLI, and Codex agents with a real-time web dashboard",
|
|
6
6
|
"workspaces": [
|
|
@@ -55,9 +55,18 @@ describe('AgentRunnerService', () => {
|
|
|
55
55
|
it('should initialize conversation state with empty messages', () => {
|
|
56
56
|
const state = runner.getState();
|
|
57
57
|
expect(state.messages).toEqual([]);
|
|
58
|
-
expect(state.systemPrompt).
|
|
58
|
+
expect(state.systemPrompt).toContain('You are a test agent.');
|
|
59
59
|
expect(state.totalTokens).toEqual({ input: 0, output: 0 });
|
|
60
60
|
});
|
|
61
|
+
|
|
62
|
+
it('appends the harness rules to every role prompt, so a weak model is told how to work', () => {
|
|
63
|
+
const prompt = runner.getState().systemPrompt;
|
|
64
|
+
expect(prompt.indexOf('You are a test agent.')).toBeLessThan(prompt.indexOf('## How to work'));
|
|
65
|
+
expect(prompt).toMatch(/never write a tool invocation as text/i);
|
|
66
|
+
expect(prompt).toMatch(/do the work in this turn/i);
|
|
67
|
+
// Naming the markup is what teaches a model to emit it.
|
|
68
|
+
expect(prompt).not.toMatch(/<\s*\/?\s*(invoke|parameter|function_calls)/i);
|
|
69
|
+
});
|
|
61
70
|
});
|
|
62
71
|
|
|
63
72
|
describe('initialize', () => {
|
|
@@ -15,11 +15,13 @@ import { createTools } from './tool-registry.js';
|
|
|
15
15
|
import { connectAndLoadMcpTools } from './mcp-tool-bridge.js';
|
|
16
16
|
import { ApprovalQueueService, type PendingApproval } from './approval-queue.service.js';
|
|
17
17
|
import { OutputFilterService } from './output-filter.service.js';
|
|
18
|
+
import { parseTextToolCalls, coerceArgs, resolveToolName, type TextToolCall, type SchemaLike } from './text-tool-calls.js';
|
|
18
19
|
import type { ToolDefinition, McpClientLike } from './types.js';
|
|
19
20
|
import {
|
|
20
21
|
type CrewlyAgentConfig,
|
|
21
22
|
type ConversationState,
|
|
22
23
|
type AgentRunResult,
|
|
24
|
+
type IncompleteReason,
|
|
23
25
|
type ToolCallRecord,
|
|
24
26
|
type CompactionResult,
|
|
25
27
|
type ContextBudgetStatus,
|
|
@@ -36,6 +38,109 @@ import {
|
|
|
36
38
|
resolveMaxOutputTokens,
|
|
37
39
|
} from './types.js';
|
|
38
40
|
|
|
41
|
+
/**
|
|
42
|
+
* What a provider's finish reason means for the turn, and how to recover.
|
|
43
|
+
*
|
|
44
|
+
* `reason === null` is the only healthy outcome; everything else left work
|
|
45
|
+
* on the table. `recoverable` turns get `budget` more attempts, each
|
|
46
|
+
* prefixed with `nudge` so the model knows to carry on rather than restart.
|
|
47
|
+
*/
|
|
48
|
+
interface FinishOutcome {
|
|
49
|
+
reason: IncompleteReason | null;
|
|
50
|
+
detail: string;
|
|
51
|
+
recoverable: boolean;
|
|
52
|
+
budget: number;
|
|
53
|
+
nudge: string;
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/** Nudge used when the model ran out of output tokens mid-answer. */
|
|
57
|
+
const CONTINUE_NUDGE =
|
|
58
|
+
'Your previous message was cut off because it hit the output limit. Continue from exactly where you stopped. Do not repeat what you already wrote, and do not start over.';
|
|
59
|
+
|
|
60
|
+
/** Nudge used after the provider ended a turn abnormally. */
|
|
61
|
+
const RESUME_NUDGE =
|
|
62
|
+
'Your previous turn ended unexpectedly before the work was finished. Review what you had already done, then carry on and complete the task. Do not repeat completed steps.';
|
|
63
|
+
|
|
64
|
+
/**
|
|
65
|
+
* Classify how a turn ended.
|
|
66
|
+
*
|
|
67
|
+
* @param finishReason - Raw provider finish reason
|
|
68
|
+
* @param steps - Steps the turn consumed
|
|
69
|
+
* @param maxSteps - The configured step ceiling
|
|
70
|
+
* @returns What it means and whether to retry
|
|
71
|
+
*/
|
|
72
|
+
export function classifyFinish(finishReason: string, steps: number, maxSteps: number): FinishOutcome {
|
|
73
|
+
const healthy: FinishOutcome = { reason: null, detail: '', recoverable: false, budget: 0, nudge: '' };
|
|
74
|
+
|
|
75
|
+
// Hitting the step ceiling is never a natural end, whatever the provider
|
|
76
|
+
// then reports — the loop was stopped from the outside mid-task.
|
|
77
|
+
if (steps >= maxSteps) {
|
|
78
|
+
return {
|
|
79
|
+
reason: 'steps-exhausted',
|
|
80
|
+
detail: `Stopped after the ${maxSteps}-step ceiling with work still outstanding.`,
|
|
81
|
+
recoverable: false,
|
|
82
|
+
budget: 0,
|
|
83
|
+
nudge: '',
|
|
84
|
+
};
|
|
85
|
+
}
|
|
86
|
+
|
|
87
|
+
switch (finishReason) {
|
|
88
|
+
case 'stop':
|
|
89
|
+
case 'tool-calls':
|
|
90
|
+
return healthy;
|
|
91
|
+
case 'length':
|
|
92
|
+
return {
|
|
93
|
+
reason: 'truncated',
|
|
94
|
+
detail: 'The model ran out of output tokens before finishing.',
|
|
95
|
+
recoverable: true,
|
|
96
|
+
budget: CREWLY_AGENT_DEFAULTS.MAX_CONTINUATIONS,
|
|
97
|
+
nudge: CONTINUE_NUDGE,
|
|
98
|
+
};
|
|
99
|
+
case 'content-filter':
|
|
100
|
+
return {
|
|
101
|
+
reason: 'content-filter',
|
|
102
|
+
detail: 'The provider refused to complete this turn on content grounds.',
|
|
103
|
+
recoverable: false,
|
|
104
|
+
budget: 0,
|
|
105
|
+
nudge: '',
|
|
106
|
+
};
|
|
107
|
+
default:
|
|
108
|
+
// 'other' | 'error' | 'unknown' | anything a provider invents.
|
|
109
|
+
return {
|
|
110
|
+
reason: 'abnormal-finish',
|
|
111
|
+
detail: `The provider ended the turn with "${finishReason}" before the work was finished.`,
|
|
112
|
+
recoverable: true,
|
|
113
|
+
budget: CREWLY_AGENT_DEFAULTS.MAX_ABNORMAL_RETRIES,
|
|
114
|
+
nudge: RESUME_NUDGE,
|
|
115
|
+
};
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
|
|
119
|
+
/**
|
|
120
|
+
* Fold a recovery attempt into the run it continues: text is appended (the
|
|
121
|
+
* model was told not to repeat itself), counters accumulate, and the newer
|
|
122
|
+
* finish reason wins.
|
|
123
|
+
*
|
|
124
|
+
* @param first - The run so far
|
|
125
|
+
* @param next - The continuation
|
|
126
|
+
* @returns The combined run
|
|
127
|
+
*/
|
|
128
|
+
export function mergeRuns(first: AgentRunResult, next: AgentRunResult): AgentRunResult {
|
|
129
|
+
const text = [first.text, next.text].map((t) => (t ?? '').trim()).filter(Boolean).join('\n\n');
|
|
130
|
+
return {
|
|
131
|
+
...next,
|
|
132
|
+
text,
|
|
133
|
+
steps: first.steps + next.steps,
|
|
134
|
+
usage: {
|
|
135
|
+
input: first.usage.input + next.usage.input,
|
|
136
|
+
output: first.usage.output + next.usage.output,
|
|
137
|
+
},
|
|
138
|
+
toolCalls: [...first.toolCalls, ...next.toolCalls],
|
|
139
|
+
budgetWarning: next.budgetWarning ?? first.budgetWarning,
|
|
140
|
+
reasoning: next.reasoning ?? first.reasoning,
|
|
141
|
+
};
|
|
142
|
+
}
|
|
143
|
+
|
|
39
144
|
/**
|
|
40
145
|
* No-op stubs for OSS-internal services. In OSS these resolve to concrete
|
|
41
146
|
* implementations (tracing, memory flush, MCP client, Slack ID synth). The
|
|
@@ -378,6 +483,25 @@ export class AgentRunnerService {
|
|
|
378
483
|
* @param modelManager - Optional model manager instance (for testing)
|
|
379
484
|
* @param apiClient - Optional API client instance (for testing)
|
|
380
485
|
*/
|
|
486
|
+
/**
|
|
487
|
+
* Rules about *how* to work that every role prompt gets, whatever the model.
|
|
488
|
+
*
|
|
489
|
+
* These exist because a weaker model fails in ways a strong one does not:
|
|
490
|
+
* it writes its tool call as prose (so nothing runs and the user sees
|
|
491
|
+
* markup), or it narrates a plan and stops without carrying it out. The
|
|
492
|
+
* runtime recovers from both, but saying so plainly costs a few tokens and
|
|
493
|
+
* prevents most of it. The wording deliberately never shows the markup
|
|
494
|
+
* syntax — describing it is what teaches a model to emit it.
|
|
495
|
+
*/
|
|
496
|
+
private static readonly HARNESS_RULES = [
|
|
497
|
+
'## How to work',
|
|
498
|
+
'',
|
|
499
|
+
'- Call tools through the tool-calling mechanism. Never write a tool invocation as text: your text is shown to the user verbatim and executes nothing.',
|
|
500
|
+
'- Do the work in this turn. If you say you will do something, do it before you finish — a plan with no action is a failed turn.',
|
|
501
|
+
'- Act, then check. Run the tool, read the result, and continue from what it actually returned rather than from what you expected.',
|
|
502
|
+
'- If something blocks you, say what blocked you and what you tried. Never report work as done that you did not verify.',
|
|
503
|
+
].join('\n');
|
|
504
|
+
|
|
381
505
|
constructor(
|
|
382
506
|
config: CrewlyAgentConfig,
|
|
383
507
|
modelManager?: ModelManager,
|
|
@@ -391,9 +515,10 @@ export class AgentRunnerService {
|
|
|
391
515
|
);
|
|
392
516
|
this.securityPolicy = { ...CREWLY_AGENT_DEFAULTS.SECURITY_POLICY };
|
|
393
517
|
// In eval mode, strip delegation-first instructions so agent implements directly
|
|
394
|
-
|
|
518
|
+
const rolePrompt = config.evalMode
|
|
395
519
|
? AgentRunnerService.stripDelegationInstructions(config.systemPrompt)
|
|
396
520
|
: config.systemPrompt;
|
|
521
|
+
this.effectiveSystemPrompt = `${rolePrompt}\n\n${AgentRunnerService.HARNESS_RULES}`;
|
|
397
522
|
// Conversation states are lazy-created on first access via the
|
|
398
523
|
// `state` getter, so we don't need to seed `__default__` here.
|
|
399
524
|
// The first message processed will create whichever conversation
|
|
@@ -1083,6 +1208,201 @@ export class AgentRunnerService {
|
|
|
1083
1208
|
private async executeRunWithStreamText(
|
|
1084
1209
|
tools: Record<string, unknown>,
|
|
1085
1210
|
abortSignal: AbortSignal,
|
|
1211
|
+
): Promise<AgentRunResult> {
|
|
1212
|
+
let result = await this.attemptWithSalvage(tools, abortSignal);
|
|
1213
|
+
let outcome = classifyFinish(result.finishReason, result.steps, this.config.maxSteps);
|
|
1214
|
+
let recoveryAttempts = 0;
|
|
1215
|
+
|
|
1216
|
+
// A turn only "finished" if the model chose to stop. Anything else left
|
|
1217
|
+
// the job half-done, and until 0.1.2 that fragment was returned as if it
|
|
1218
|
+
// were the answer — the agent would promise to do something, get cut off,
|
|
1219
|
+
// and the user was told it was done (2026-09-19: the orchestrator ended
|
|
1220
|
+
// 4 turns in a row on `other` and silently created nothing).
|
|
1221
|
+
while (outcome.recoverable && recoveryAttempts < outcome.budget && !abortSignal.aborted) {
|
|
1222
|
+
recoveryAttempts++;
|
|
1223
|
+
this.streamingCallbacks.onTextChunk?.(`[recover] ${outcome.reason} — continuing (${recoveryAttempts}/${outcome.budget})\n`);
|
|
1224
|
+
this.state.messages.push({ role: 'user', content: outcome.nudge });
|
|
1225
|
+
const next = await this.attemptWithSalvage(tools, abortSignal);
|
|
1226
|
+
result = mergeRuns(result, next);
|
|
1227
|
+
outcome = classifyFinish(next.finishReason, next.steps, this.config.maxSteps);
|
|
1228
|
+
}
|
|
1229
|
+
|
|
1230
|
+
if (outcome.reason === null) return result;
|
|
1231
|
+
|
|
1232
|
+
return {
|
|
1233
|
+
...result,
|
|
1234
|
+
incomplete: {
|
|
1235
|
+
reason: outcome.reason,
|
|
1236
|
+
detail: outcome.detail,
|
|
1237
|
+
finishReason: result.finishReason,
|
|
1238
|
+
recoveryAttempts,
|
|
1239
|
+
},
|
|
1240
|
+
};
|
|
1241
|
+
}
|
|
1242
|
+
|
|
1243
|
+
/**
|
|
1244
|
+
* Run one turn, then rescue any tool call the model *wrote* instead of called.
|
|
1245
|
+
*
|
|
1246
|
+
* A weak model sometimes emits its call envelope into the text channel:
|
|
1247
|
+
* the provider returns prose, the SDK sees no tool call, and the step ends
|
|
1248
|
+
* having done nothing — the failure mode behind both the markup users saw
|
|
1249
|
+
* in Slack and the turns that promised work and produced none. Rather than
|
|
1250
|
+
* strip the markup and lose the intent, the envelope is parsed, the tools
|
|
1251
|
+
* are executed for real, and the results are handed back so the turn can
|
|
1252
|
+
* carry on.
|
|
1253
|
+
*
|
|
1254
|
+
* Salvaged calls go through the tool's own `execute`, so approval gates and
|
|
1255
|
+
* command blocklists apply exactly as they do to a native call — this
|
|
1256
|
+
* recovers lost work, it does not widen what the agent may do.
|
|
1257
|
+
*
|
|
1258
|
+
* @param tools - The tool registry for this run
|
|
1259
|
+
* @param abortSignal - Cancels the turn and any further rounds
|
|
1260
|
+
* @returns The turn's result, with salvaged calls folded into `toolCalls`
|
|
1261
|
+
*/
|
|
1262
|
+
private async attemptWithSalvage(
|
|
1263
|
+
tools: Record<string, unknown>,
|
|
1264
|
+
abortSignal: AbortSignal,
|
|
1265
|
+
): Promise<AgentRunResult> {
|
|
1266
|
+
let attempt = await this.attemptWithErrorRetries(tools, abortSignal);
|
|
1267
|
+
const salvaged: ToolCallRecord[] = [];
|
|
1268
|
+
|
|
1269
|
+
for (let round = 0; round < CREWLY_AGENT_DEFAULTS.MAX_TEXT_TOOL_SALVAGES; round++) {
|
|
1270
|
+
if (abortSignal.aborted) break;
|
|
1271
|
+
const parsed = parseTextToolCalls(attempt.text ?? '');
|
|
1272
|
+
if (parsed.calls.length === 0) break;
|
|
1273
|
+
|
|
1274
|
+
console.warn('[AgentRunner] Model wrote tool calls as text — executing them:', {
|
|
1275
|
+
round: round + 1,
|
|
1276
|
+
tools: parsed.calls.map(c => c.toolName),
|
|
1277
|
+
});
|
|
1278
|
+
this.streamingCallbacks.onTextChunk?.(
|
|
1279
|
+
`[salvage] running ${parsed.calls.length} tool call(s) the model wrote as text\n`,
|
|
1280
|
+
);
|
|
1281
|
+
|
|
1282
|
+
const { records, report } = await this.runSalvagedCalls(parsed.calls, tools);
|
|
1283
|
+
salvaged.push(...records);
|
|
1284
|
+
|
|
1285
|
+
// Rewrite the turn's own message so the transcript does not keep
|
|
1286
|
+
// teaching the model that writing markup is how a tool gets called.
|
|
1287
|
+
this.replaceLastAssistantMessage(parsed.text || '(I wrote a tool call as text instead of calling the tool.)');
|
|
1288
|
+
this.state.messages.push({ role: 'user', content: report });
|
|
1289
|
+
|
|
1290
|
+
const next = await this.attemptWithErrorRetries(tools, abortSignal);
|
|
1291
|
+
attempt = mergeRuns({ ...attempt, text: parsed.text }, next);
|
|
1292
|
+
}
|
|
1293
|
+
|
|
1294
|
+
return salvaged.length > 0
|
|
1295
|
+
? { ...attempt, toolCalls: [...attempt.toolCalls, ...salvaged] }
|
|
1296
|
+
: attempt;
|
|
1297
|
+
}
|
|
1298
|
+
|
|
1299
|
+
/**
|
|
1300
|
+
* Execute the tool calls recovered from text and describe the results.
|
|
1301
|
+
*
|
|
1302
|
+
* Unknown tools and arguments the schema rejects are reported back rather
|
|
1303
|
+
* than guessed at: the model gets a specific complaint it can act on, which
|
|
1304
|
+
* is far more useful than silence.
|
|
1305
|
+
*
|
|
1306
|
+
* @param calls - Calls parsed out of the model's text
|
|
1307
|
+
* @param tools - The tool registry for this run
|
|
1308
|
+
* @returns Records for the run ledger, and the message to feed back
|
|
1309
|
+
*/
|
|
1310
|
+
private async runSalvagedCalls(
|
|
1311
|
+
calls: TextToolCall[],
|
|
1312
|
+
tools: Record<string, unknown>,
|
|
1313
|
+
): Promise<{ records: ToolCallRecord[]; report: string }> {
|
|
1314
|
+
const records: ToolCallRecord[] = [];
|
|
1315
|
+
const sections: string[] = [];
|
|
1316
|
+
const registry = tools as Record<string, ToolDefinition | undefined>;
|
|
1317
|
+
|
|
1318
|
+
for (const call of calls.slice(0, CREWLY_AGENT_DEFAULTS.MAX_SALVAGED_CALLS_PER_ROUND)) {
|
|
1319
|
+
// A model that has seen another harness asks for `Bash`, not `bash_exec`.
|
|
1320
|
+
const resolved = resolveToolName(call.toolName, Object.keys(registry));
|
|
1321
|
+
const def = resolved ? registry[resolved] : undefined;
|
|
1322
|
+
if (!resolved || !def || typeof def.execute !== 'function') {
|
|
1323
|
+
sections.push(`### ${call.toolName}\nThere is no tool with that name. Available tools: ${Object.keys(registry).join(', ')}`);
|
|
1324
|
+
continue;
|
|
1325
|
+
}
|
|
1326
|
+
|
|
1327
|
+
const { args, error } = coerceArgs(call.args, def.inputSchema as unknown as SchemaLike | undefined);
|
|
1328
|
+
if (error) {
|
|
1329
|
+
sections.push(`### ${resolved}\nThe arguments were rejected: ${error}`);
|
|
1330
|
+
continue;
|
|
1331
|
+
}
|
|
1332
|
+
|
|
1333
|
+
const startedAt = Date.now();
|
|
1334
|
+
this.streamingCallbacks.onToolCallStart?.(resolved, args);
|
|
1335
|
+
let output: unknown;
|
|
1336
|
+
try {
|
|
1337
|
+
output = await def.execute(args);
|
|
1338
|
+
} catch (err) {
|
|
1339
|
+
output = { error: err instanceof Error ? err.message : String(err) };
|
|
1340
|
+
}
|
|
1341
|
+
this.streamingCallbacks.onToolCallFinish?.(resolved, args, output, Date.now() - startedAt);
|
|
1342
|
+
|
|
1343
|
+
records.push({ toolName: resolved, args, result: output });
|
|
1344
|
+
// Name it as the model wrote it when that differed, so it learns the real name.
|
|
1345
|
+
const heading = resolved === call.toolName ? resolved : `${resolved} (you wrote "${call.toolName}")`;
|
|
1346
|
+
sections.push(`### ${heading}\n${this.summarizeSalvagedResult(output)}`);
|
|
1347
|
+
}
|
|
1348
|
+
|
|
1349
|
+
const skipped = calls.length - Math.min(calls.length, CREWLY_AGENT_DEFAULTS.MAX_SALVAGED_CALLS_PER_ROUND);
|
|
1350
|
+
const report = [
|
|
1351
|
+
'You wrote your tool calls as text, so the model API never received them. I executed them for you; here is what they returned.',
|
|
1352
|
+
...sections,
|
|
1353
|
+
skipped > 0 ? `(${skipped} further call(s) were not run — make them yourself.)` : '',
|
|
1354
|
+
'Use the tool-calling mechanism from now on: text in your reply is shown to the user verbatim and executes nothing. Continue the task with these results.',
|
|
1355
|
+
].filter(Boolean).join('\n\n');
|
|
1356
|
+
|
|
1357
|
+
return { records, report };
|
|
1358
|
+
}
|
|
1359
|
+
|
|
1360
|
+
/**
|
|
1361
|
+
* Render a salvaged tool result small enough to feed back.
|
|
1362
|
+
*
|
|
1363
|
+
* @param output - Whatever the tool returned
|
|
1364
|
+
* @returns A string, truncated with a note when it was long
|
|
1365
|
+
*/
|
|
1366
|
+
private summarizeSalvagedResult(output: unknown): string {
|
|
1367
|
+
let rendered: string;
|
|
1368
|
+
try {
|
|
1369
|
+
rendered = typeof output === 'string' ? output : JSON.stringify(output, null, 2) ?? String(output);
|
|
1370
|
+
} catch {
|
|
1371
|
+
rendered = String(output);
|
|
1372
|
+
}
|
|
1373
|
+
const limit = CREWLY_AGENT_DEFAULTS.SALVAGED_RESULT_MAX_CHARS;
|
|
1374
|
+
return rendered.length > limit
|
|
1375
|
+
? `${rendered.slice(0, limit)}\n… (truncated, ${rendered.length - limit} more characters)`
|
|
1376
|
+
: rendered;
|
|
1377
|
+
}
|
|
1378
|
+
|
|
1379
|
+
/**
|
|
1380
|
+
* Rewrite the assistant message this turn just added to the transcript.
|
|
1381
|
+
*
|
|
1382
|
+
* Stops at the user message that opened the turn, so an earlier, healthy
|
|
1383
|
+
* reply is never touched.
|
|
1384
|
+
*
|
|
1385
|
+
* @param content - Replacement content
|
|
1386
|
+
*/
|
|
1387
|
+
private replaceLastAssistantMessage(content: string): void {
|
|
1388
|
+
for (let i = this.state.messages.length - 1; i >= 0; i--) {
|
|
1389
|
+
const message = this.state.messages[i];
|
|
1390
|
+
if (message.role === 'assistant') {
|
|
1391
|
+
this.state.messages[i] = { ...message, content };
|
|
1392
|
+
return;
|
|
1393
|
+
}
|
|
1394
|
+
if (message.role === 'user') return;
|
|
1395
|
+
}
|
|
1396
|
+
}
|
|
1397
|
+
|
|
1398
|
+
/**
|
|
1399
|
+
* Run one turn, retrying only on *thrown* failures (rate limits, network,
|
|
1400
|
+
* context length). A turn that returns with a bad `finishReason` is the
|
|
1401
|
+
* caller's problem — see {@link executeRunWithStreamText}.
|
|
1402
|
+
*/
|
|
1403
|
+
private async attemptWithErrorRetries(
|
|
1404
|
+
tools: Record<string, unknown>,
|
|
1405
|
+
abortSignal: AbortSignal,
|
|
1086
1406
|
): Promise<AgentRunResult> {
|
|
1087
1407
|
const maxRetries = CREWLY_AGENT_DEFAULTS.MAX_RETRIES;
|
|
1088
1408
|
const baseDelay = CREWLY_AGENT_DEFAULTS.RETRY_BASE_DELAY_MS;
|
|
@@ -0,0 +1,177 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Tests for how a turn that did not end naturally is classified, continued
|
|
3
|
+
* and — when it cannot be rescued — reported as incomplete.
|
|
4
|
+
*
|
|
5
|
+
* The bug these lock down: a turn ending on `length` or `other` used to be
|
|
6
|
+
* returned as a finished answer, so the agent could promise work, get cut
|
|
7
|
+
* off, and report success (2026-09-19, the orchestrator ended four turns in
|
|
8
|
+
* a row on `other` and silently created nothing).
|
|
9
|
+
*/
|
|
10
|
+
|
|
11
|
+
import { describe, it, expect, vi, beforeEach } from 'vitest';
|
|
12
|
+
import { AgentRunnerService, classifyFinish, mergeRuns } from './agent-runner.service.js';
|
|
13
|
+
import { CREWLY_AGENT_DEFAULTS, type AgentRunResult, type CrewlyAgentConfig } from './types.js';
|
|
14
|
+
|
|
15
|
+
const MAX_STEPS = 50;
|
|
16
|
+
|
|
17
|
+
/** A run result with sensible defaults. */
|
|
18
|
+
function runResult(over: Partial<AgentRunResult> = {}): AgentRunResult {
|
|
19
|
+
return {
|
|
20
|
+
text: 'text',
|
|
21
|
+
steps: 1,
|
|
22
|
+
usage: { input: 10, output: 5 },
|
|
23
|
+
toolCalls: [],
|
|
24
|
+
finishReason: 'stop',
|
|
25
|
+
...over,
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
describe('classifyFinish', () => {
|
|
30
|
+
it('treats stop and tool-calls as the only healthy endings', () => {
|
|
31
|
+
for (const reason of ['stop', 'tool-calls']) {
|
|
32
|
+
expect(classifyFinish(reason, 3, MAX_STEPS)).toMatchObject({ reason: null, recoverable: false });
|
|
33
|
+
}
|
|
34
|
+
});
|
|
35
|
+
|
|
36
|
+
it('continues a turn cut off by the output limit, within the continuation budget', () => {
|
|
37
|
+
const out = classifyFinish('length', 3, MAX_STEPS);
|
|
38
|
+
expect(out).toMatchObject({ reason: 'truncated', recoverable: true, budget: CREWLY_AGENT_DEFAULTS.MAX_CONTINUATIONS });
|
|
39
|
+
expect(out.nudge).toMatch(/continue from exactly where you stopped/i);
|
|
40
|
+
expect(out.nudge).toMatch(/do not repeat/i);
|
|
41
|
+
});
|
|
42
|
+
|
|
43
|
+
it('retries an abnormal provider finish once, naming the reason', () => {
|
|
44
|
+
for (const reason of ['other', 'error', 'unknown', 'insufficient_system_resource']) {
|
|
45
|
+
const out = classifyFinish(reason, 3, MAX_STEPS);
|
|
46
|
+
expect(out).toMatchObject({ reason: 'abnormal-finish', recoverable: true, budget: CREWLY_AGENT_DEFAULTS.MAX_ABNORMAL_RETRIES });
|
|
47
|
+
expect(out.detail).toContain(reason);
|
|
48
|
+
}
|
|
49
|
+
});
|
|
50
|
+
|
|
51
|
+
it('never retries a content-filter refusal', () => {
|
|
52
|
+
expect(classifyFinish('content-filter', 3, MAX_STEPS)).toMatchObject({ reason: 'content-filter', recoverable: false, budget: 0 });
|
|
53
|
+
});
|
|
54
|
+
|
|
55
|
+
it('reports the step ceiling as incomplete whatever the provider says, and does not retry into it', () => {
|
|
56
|
+
// Even a 'stop' at the ceiling means the loop was cut from outside.
|
|
57
|
+
expect(classifyFinish('stop', MAX_STEPS, MAX_STEPS)).toMatchObject({ reason: 'steps-exhausted', recoverable: false });
|
|
58
|
+
expect(classifyFinish('length', MAX_STEPS + 2, MAX_STEPS)).toMatchObject({ reason: 'steps-exhausted' });
|
|
59
|
+
expect(classifyFinish('stop', MAX_STEPS - 1, MAX_STEPS)).toMatchObject({ reason: null });
|
|
60
|
+
});
|
|
61
|
+
});
|
|
62
|
+
|
|
63
|
+
describe('mergeRuns', () => {
|
|
64
|
+
it('appends text and accumulates steps, usage and tool calls', () => {
|
|
65
|
+
const merged = mergeRuns(
|
|
66
|
+
runResult({ text: 'part one', steps: 4, usage: { input: 10, output: 20 }, toolCalls: [{ toolName: 'a', args: {}, result: 1 }], finishReason: 'length' }),
|
|
67
|
+
runResult({ text: 'part two', steps: 2, usage: { input: 3, output: 7 }, toolCalls: [{ toolName: 'b', args: {}, result: 2 }], finishReason: 'stop' }),
|
|
68
|
+
);
|
|
69
|
+
expect(merged.text).toBe('part one\n\npart two');
|
|
70
|
+
expect(merged.steps).toBe(6);
|
|
71
|
+
expect(merged.usage).toEqual({ input: 13, output: 27 });
|
|
72
|
+
expect(merged.toolCalls.map((t) => t.toolName)).toEqual(['a', 'b']);
|
|
73
|
+
expect(merged.finishReason).toBe('stop');
|
|
74
|
+
});
|
|
75
|
+
|
|
76
|
+
it('drops empty halves rather than leaving blank gaps, and keeps the newer metadata', () => {
|
|
77
|
+
expect(mergeRuns(runResult({ text: '' }), runResult({ text: 'only' })).text).toBe('only');
|
|
78
|
+
expect(mergeRuns(runResult({ text: 'only' }), runResult({ text: ' ' })).text).toBe('only');
|
|
79
|
+
const merged = mergeRuns(runResult({ budgetWarning: 'old', reasoning: 'r1' }), runResult({ budgetWarning: undefined, reasoning: null }));
|
|
80
|
+
expect(merged.budgetWarning).toBe('old');
|
|
81
|
+
expect(merged.reasoning).toBe('r1');
|
|
82
|
+
});
|
|
83
|
+
});
|
|
84
|
+
|
|
85
|
+
describe('recovery loop', () => {
|
|
86
|
+
let runner: AgentRunnerService;
|
|
87
|
+
let attempts: AgentRunResult[];
|
|
88
|
+
let attemptSpy: ReturnType<typeof vi.fn>;
|
|
89
|
+
|
|
90
|
+
const config = {
|
|
91
|
+
sessionName: 'test-agent',
|
|
92
|
+
role: 'developer',
|
|
93
|
+
projectPath: '/tmp',
|
|
94
|
+
maxSteps: MAX_STEPS,
|
|
95
|
+
model: { provider: 'deepseek', modelId: 'deepseek-chat' },
|
|
96
|
+
} as unknown as CrewlyAgentConfig;
|
|
97
|
+
|
|
98
|
+
/** Drive the private recovery loop with a scripted sequence of attempts. */
|
|
99
|
+
async function runLoop(): Promise<AgentRunResult> {
|
|
100
|
+
return await (runner as unknown as {
|
|
101
|
+
executeRunWithStreamText(tools: Record<string, unknown>, signal: AbortSignal): Promise<AgentRunResult>;
|
|
102
|
+
}).executeRunWithStreamText({}, new AbortController().signal);
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
beforeEach(() => {
|
|
106
|
+
runner = new AgentRunnerService(config);
|
|
107
|
+
attempts = [];
|
|
108
|
+
attemptSpy = vi.fn(async () => attempts.shift() ?? runResult());
|
|
109
|
+
(runner as unknown as Record<string, unknown>).attemptWithErrorRetries = attemptSpy;
|
|
110
|
+
});
|
|
111
|
+
|
|
112
|
+
it('returns a healthy turn untouched, with no extra model call', async () => {
|
|
113
|
+
attempts = [runResult({ text: 'done', finishReason: 'stop' })];
|
|
114
|
+
const out = await runLoop();
|
|
115
|
+
expect(out.text).toBe('done');
|
|
116
|
+
expect(out.incomplete).toBeUndefined();
|
|
117
|
+
expect(attemptSpy).toHaveBeenCalledTimes(1);
|
|
118
|
+
});
|
|
119
|
+
|
|
120
|
+
it('continues a truncated turn until the model stops, and returns the joined answer', async () => {
|
|
121
|
+
attempts = [
|
|
122
|
+
runResult({ text: 'first half', finishReason: 'length' }),
|
|
123
|
+
runResult({ text: 'second half', finishReason: 'stop' }),
|
|
124
|
+
];
|
|
125
|
+
const out = await runLoop();
|
|
126
|
+
expect(out.text).toBe('first half\n\nsecond half');
|
|
127
|
+
expect(out.incomplete).toBeUndefined();
|
|
128
|
+
expect(attemptSpy).toHaveBeenCalledTimes(2);
|
|
129
|
+
});
|
|
130
|
+
|
|
131
|
+
it('gives up after the continuation budget and reports what it kept', async () => {
|
|
132
|
+
attempts = Array.from({ length: 10 }, (_, i) => runResult({ text: `chunk ${i}`, finishReason: 'length' }));
|
|
133
|
+
const out = await runLoop();
|
|
134
|
+
expect(attemptSpy).toHaveBeenCalledTimes(CREWLY_AGENT_DEFAULTS.MAX_CONTINUATIONS + 1);
|
|
135
|
+
expect(out.incomplete).toMatchObject({ reason: 'truncated', recoveryAttempts: CREWLY_AGENT_DEFAULTS.MAX_CONTINUATIONS });
|
|
136
|
+
expect(out.text).toContain('chunk 0');
|
|
137
|
+
});
|
|
138
|
+
|
|
139
|
+
it('recovers the deepseek case: an abnormal finish retried once that then completes', async () => {
|
|
140
|
+
attempts = [
|
|
141
|
+
runResult({ text: "I'll create the team", finishReason: 'other' }),
|
|
142
|
+
runResult({ text: 'Team created.', finishReason: 'stop' }),
|
|
143
|
+
];
|
|
144
|
+
const out = await runLoop();
|
|
145
|
+
expect(out.incomplete).toBeUndefined();
|
|
146
|
+
expect(out.text).toBe("I'll create the team\n\nTeam created.");
|
|
147
|
+
});
|
|
148
|
+
|
|
149
|
+
it('marks the turn incomplete when the provider keeps bailing', async () => {
|
|
150
|
+
attempts = [
|
|
151
|
+
runResult({ text: "I'll create the team", finishReason: 'other' }),
|
|
152
|
+
runResult({ text: '', finishReason: 'other' }),
|
|
153
|
+
];
|
|
154
|
+
const out = await runLoop();
|
|
155
|
+
expect(attemptSpy).toHaveBeenCalledTimes(2);
|
|
156
|
+
expect(out.incomplete).toMatchObject({ reason: 'abnormal-finish', finishReason: 'other', recoveryAttempts: 1 });
|
|
157
|
+
});
|
|
158
|
+
|
|
159
|
+
it('does not retry a step-exhausted or content-filtered turn', async () => {
|
|
160
|
+
attempts = [runResult({ steps: MAX_STEPS, finishReason: 'tool-calls' })];
|
|
161
|
+
expect((await runLoop()).incomplete).toMatchObject({ reason: 'steps-exhausted', recoveryAttempts: 0 });
|
|
162
|
+
expect(attemptSpy).toHaveBeenCalledTimes(1);
|
|
163
|
+
|
|
164
|
+
attemptSpy.mockClear();
|
|
165
|
+
attempts = [runResult({ finishReason: 'content-filter' })];
|
|
166
|
+
expect((await runLoop()).incomplete).toMatchObject({ reason: 'content-filter' });
|
|
167
|
+
expect(attemptSpy).toHaveBeenCalledTimes(1);
|
|
168
|
+
});
|
|
169
|
+
|
|
170
|
+
it('pushes a continuation instruction into the conversation before retrying', async () => {
|
|
171
|
+
attempts = [runResult({ finishReason: 'length' }), runResult({ finishReason: 'stop' })];
|
|
172
|
+
await runLoop();
|
|
173
|
+
const messages = (runner as unknown as { state: { messages: Array<{ role: string; content: string }> } }).state.messages;
|
|
174
|
+
const nudge = messages.filter((m) => m.role === 'user').pop();
|
|
175
|
+
expect(nudge?.content).toMatch(/continue from exactly where you stopped/i);
|
|
176
|
+
});
|
|
177
|
+
});
|