crewly 1.20.34 → 1.20.35

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -1,4 +1,4 @@
1
1
  {
2
- "commit": "82d2b99f83607a55b36a0982c9463e1d65c85f19",
3
- "builtAt": "2026-09-19T23:52:29.078Z"
2
+ "commit": "8cc20b94b33b310ecb16ebc59b6aec4d86651d86",
3
+ "builtAt": "2026-09-19T23:54:35.448Z"
4
4
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "crewly",
3
- "version": "1.20.34",
3
+ "version": "1.20.35",
4
4
  "type": "module",
5
5
  "description": "Multi-agent orchestration platform for AI coding teams — coordinates Claude Code, Gemini CLI, and Codex agents with a real-time web dashboard",
6
6
  "workspaces": [
@@ -20,6 +20,7 @@ import {
20
20
  type CrewlyAgentConfig,
21
21
  type ConversationState,
22
22
  type AgentRunResult,
23
+ type IncompleteReason,
23
24
  type ToolCallRecord,
24
25
  type CompactionResult,
25
26
  type ContextBudgetStatus,
@@ -36,6 +37,109 @@ import {
36
37
  resolveMaxOutputTokens,
37
38
  } from './types.js';
38
39
 
40
+ /**
41
+ * What a provider's finish reason means for the turn, and how to recover.
42
+ *
43
+ * `reason === null` is the only healthy outcome; everything else left work
44
+ * on the table. `recoverable` turns get `budget` more attempts, each
45
+ * prefixed with `nudge` so the model knows to carry on rather than restart.
46
+ */
47
+ interface FinishOutcome {
48
+ reason: IncompleteReason | null;
49
+ detail: string;
50
+ recoverable: boolean;
51
+ budget: number;
52
+ nudge: string;
53
+ }
54
+
55
+ /** Nudge used when the model ran out of output tokens mid-answer. */
56
+ const CONTINUE_NUDGE =
57
+ 'Your previous message was cut off because it hit the output limit. Continue from exactly where you stopped. Do not repeat what you already wrote, and do not start over.';
58
+
59
+ /** Nudge used after the provider ended a turn abnormally. */
60
+ const RESUME_NUDGE =
61
+ 'Your previous turn ended unexpectedly before the work was finished. Review what you had already done, then carry on and complete the task. Do not repeat completed steps.';
62
+
63
+ /**
64
+ * Classify how a turn ended.
65
+ *
66
+ * @param finishReason - Raw provider finish reason
67
+ * @param steps - Steps the turn consumed
68
+ * @param maxSteps - The configured step ceiling
69
+ * @returns What it means and whether to retry
70
+ */
71
+ export function classifyFinish(finishReason: string, steps: number, maxSteps: number): FinishOutcome {
72
+ const healthy: FinishOutcome = { reason: null, detail: '', recoverable: false, budget: 0, nudge: '' };
73
+
74
+ // Hitting the step ceiling is never a natural end, whatever the provider
75
+ // then reports — the loop was stopped from the outside mid-task.
76
+ if (steps >= maxSteps) {
77
+ return {
78
+ reason: 'steps-exhausted',
79
+ detail: `Stopped after the ${maxSteps}-step ceiling with work still outstanding.`,
80
+ recoverable: false,
81
+ budget: 0,
82
+ nudge: '',
83
+ };
84
+ }
85
+
86
+ switch (finishReason) {
87
+ case 'stop':
88
+ case 'tool-calls':
89
+ return healthy;
90
+ case 'length':
91
+ return {
92
+ reason: 'truncated',
93
+ detail: 'The model ran out of output tokens before finishing.',
94
+ recoverable: true,
95
+ budget: CREWLY_AGENT_DEFAULTS.MAX_CONTINUATIONS,
96
+ nudge: CONTINUE_NUDGE,
97
+ };
98
+ case 'content-filter':
99
+ return {
100
+ reason: 'content-filter',
101
+ detail: 'The provider refused to complete this turn on content grounds.',
102
+ recoverable: false,
103
+ budget: 0,
104
+ nudge: '',
105
+ };
106
+ default:
107
+ // 'other' | 'error' | 'unknown' | anything a provider invents.
108
+ return {
109
+ reason: 'abnormal-finish',
110
+ detail: `The provider ended the turn with "${finishReason}" before the work was finished.`,
111
+ recoverable: true,
112
+ budget: CREWLY_AGENT_DEFAULTS.MAX_ABNORMAL_RETRIES,
113
+ nudge: RESUME_NUDGE,
114
+ };
115
+ }
116
+ }
117
+
118
+ /**
119
+ * Fold a recovery attempt into the run it continues: text is appended (the
120
+ * model was told not to repeat itself), counters accumulate, and the newer
121
+ * finish reason wins.
122
+ *
123
+ * @param first - The run so far
124
+ * @param next - The continuation
125
+ * @returns The combined run
126
+ */
127
+ export function mergeRuns(first: AgentRunResult, next: AgentRunResult): AgentRunResult {
128
+ const text = [first.text, next.text].map((t) => (t ?? '').trim()).filter(Boolean).join('\n\n');
129
+ return {
130
+ ...next,
131
+ text,
132
+ steps: first.steps + next.steps,
133
+ usage: {
134
+ input: first.usage.input + next.usage.input,
135
+ output: first.usage.output + next.usage.output,
136
+ },
137
+ toolCalls: [...first.toolCalls, ...next.toolCalls],
138
+ budgetWarning: next.budgetWarning ?? first.budgetWarning,
139
+ reasoning: next.reasoning ?? first.reasoning,
140
+ };
141
+ }
142
+
39
143
  /**
40
144
  * No-op stubs for OSS-internal services. In OSS these resolve to concrete
41
145
  * implementations (tracing, memory flush, MCP client, Slack ID synth). The
@@ -1083,6 +1187,46 @@ export class AgentRunnerService {
1083
1187
  private async executeRunWithStreamText(
1084
1188
  tools: Record<string, unknown>,
1085
1189
  abortSignal: AbortSignal,
1190
+ ): Promise<AgentRunResult> {
1191
+ let result = await this.attemptWithErrorRetries(tools, abortSignal);
1192
+ let outcome = classifyFinish(result.finishReason, result.steps, this.config.maxSteps);
1193
+ let recoveryAttempts = 0;
1194
+
1195
+ // A turn only "finished" if the model chose to stop. Anything else left
1196
+ // the job half-done, and until 0.1.2 that fragment was returned as if it
1197
+ // were the answer — the agent would promise to do something, get cut off,
1198
+ // and the user was told it was done (2026-09-19: the orchestrator ended
1199
+ // 4 turns in a row on `other` and silently created nothing).
1200
+ while (outcome.recoverable && recoveryAttempts < outcome.budget && !abortSignal.aborted) {
1201
+ recoveryAttempts++;
1202
+ this.streamingCallbacks.onTextChunk?.(`[recover] ${outcome.reason} — continuing (${recoveryAttempts}/${outcome.budget})\n`);
1203
+ this.state.messages.push({ role: 'user', content: outcome.nudge });
1204
+ const next = await this.attemptWithErrorRetries(tools, abortSignal);
1205
+ result = mergeRuns(result, next);
1206
+ outcome = classifyFinish(next.finishReason, next.steps, this.config.maxSteps);
1207
+ }
1208
+
1209
+ if (outcome.reason === null) return result;
1210
+
1211
+ return {
1212
+ ...result,
1213
+ incomplete: {
1214
+ reason: outcome.reason,
1215
+ detail: outcome.detail,
1216
+ finishReason: result.finishReason,
1217
+ recoveryAttempts,
1218
+ },
1219
+ };
1220
+ }
1221
+
1222
+ /**
1223
+ * Run one turn, retrying only on *thrown* failures (rate limits, network,
1224
+ * context length). A turn that returns with a bad `finishReason` is the
1225
+ * caller's problem — see {@link executeRunWithStreamText}.
1226
+ */
1227
+ private async attemptWithErrorRetries(
1228
+ tools: Record<string, unknown>,
1229
+ abortSignal: AbortSignal,
1086
1230
  ): Promise<AgentRunResult> {
1087
1231
  const maxRetries = CREWLY_AGENT_DEFAULTS.MAX_RETRIES;
1088
1232
  const baseDelay = CREWLY_AGENT_DEFAULTS.RETRY_BASE_DELAY_MS;
@@ -0,0 +1,177 @@
1
+ /**
2
+ * Tests for how a turn that did not end naturally is classified, continued
3
+ * and — when it cannot be rescued — reported as incomplete.
4
+ *
5
+ * The bug these lock down: a turn ending on `length` or `other` used to be
6
+ * returned as a finished answer, so the agent could promise work, get cut
7
+ * off, and report success (2026-09-19, the orchestrator ended four turns in
8
+ * a row on `other` and silently created nothing).
9
+ */
10
+
11
+ import { describe, it, expect, vi, beforeEach } from 'vitest';
12
+ import { AgentRunnerService, classifyFinish, mergeRuns } from './agent-runner.service.js';
13
+ import { CREWLY_AGENT_DEFAULTS, type AgentRunResult, type CrewlyAgentConfig } from './types.js';
14
+
15
+ const MAX_STEPS = 50;
16
+
17
+ /** A run result with sensible defaults. */
18
+ function runResult(over: Partial<AgentRunResult> = {}): AgentRunResult {
19
+ return {
20
+ text: 'text',
21
+ steps: 1,
22
+ usage: { input: 10, output: 5 },
23
+ toolCalls: [],
24
+ finishReason: 'stop',
25
+ ...over,
26
+ };
27
+ }
28
+
29
+ describe('classifyFinish', () => {
30
+ it('treats stop and tool-calls as the only healthy endings', () => {
31
+ for (const reason of ['stop', 'tool-calls']) {
32
+ expect(classifyFinish(reason, 3, MAX_STEPS)).toMatchObject({ reason: null, recoverable: false });
33
+ }
34
+ });
35
+
36
+ it('continues a turn cut off by the output limit, within the continuation budget', () => {
37
+ const out = classifyFinish('length', 3, MAX_STEPS);
38
+ expect(out).toMatchObject({ reason: 'truncated', recoverable: true, budget: CREWLY_AGENT_DEFAULTS.MAX_CONTINUATIONS });
39
+ expect(out.nudge).toMatch(/continue from exactly where you stopped/i);
40
+ expect(out.nudge).toMatch(/do not repeat/i);
41
+ });
42
+
43
+ it('retries an abnormal provider finish once, naming the reason', () => {
44
+ for (const reason of ['other', 'error', 'unknown', 'insufficient_system_resource']) {
45
+ const out = classifyFinish(reason, 3, MAX_STEPS);
46
+ expect(out).toMatchObject({ reason: 'abnormal-finish', recoverable: true, budget: CREWLY_AGENT_DEFAULTS.MAX_ABNORMAL_RETRIES });
47
+ expect(out.detail).toContain(reason);
48
+ }
49
+ });
50
+
51
+ it('never retries a content-filter refusal', () => {
52
+ expect(classifyFinish('content-filter', 3, MAX_STEPS)).toMatchObject({ reason: 'content-filter', recoverable: false, budget: 0 });
53
+ });
54
+
55
+ it('reports the step ceiling as incomplete whatever the provider says, and does not retry into it', () => {
56
+ // Even a 'stop' at the ceiling means the loop was cut from outside.
57
+ expect(classifyFinish('stop', MAX_STEPS, MAX_STEPS)).toMatchObject({ reason: 'steps-exhausted', recoverable: false });
58
+ expect(classifyFinish('length', MAX_STEPS + 2, MAX_STEPS)).toMatchObject({ reason: 'steps-exhausted' });
59
+ expect(classifyFinish('stop', MAX_STEPS - 1, MAX_STEPS)).toMatchObject({ reason: null });
60
+ });
61
+ });
62
+
63
+ describe('mergeRuns', () => {
64
+ it('appends text and accumulates steps, usage and tool calls', () => {
65
+ const merged = mergeRuns(
66
+ runResult({ text: 'part one', steps: 4, usage: { input: 10, output: 20 }, toolCalls: [{ toolName: 'a', args: {}, result: 1 }], finishReason: 'length' }),
67
+ runResult({ text: 'part two', steps: 2, usage: { input: 3, output: 7 }, toolCalls: [{ toolName: 'b', args: {}, result: 2 }], finishReason: 'stop' }),
68
+ );
69
+ expect(merged.text).toBe('part one\n\npart two');
70
+ expect(merged.steps).toBe(6);
71
+ expect(merged.usage).toEqual({ input: 13, output: 27 });
72
+ expect(merged.toolCalls.map((t) => t.toolName)).toEqual(['a', 'b']);
73
+ expect(merged.finishReason).toBe('stop');
74
+ });
75
+
76
+ it('drops empty halves rather than leaving blank gaps, and keeps the newer metadata', () => {
77
+ expect(mergeRuns(runResult({ text: '' }), runResult({ text: 'only' })).text).toBe('only');
78
+ expect(mergeRuns(runResult({ text: 'only' }), runResult({ text: ' ' })).text).toBe('only');
79
+ const merged = mergeRuns(runResult({ budgetWarning: 'old', reasoning: 'r1' }), runResult({ budgetWarning: undefined, reasoning: null }));
80
+ expect(merged.budgetWarning).toBe('old');
81
+ expect(merged.reasoning).toBe('r1');
82
+ });
83
+ });
84
+
85
+ describe('recovery loop', () => {
86
+ let runner: AgentRunnerService;
87
+ let attempts: AgentRunResult[];
88
+ let attemptSpy: ReturnType<typeof vi.fn>;
89
+
90
+ const config = {
91
+ sessionName: 'test-agent',
92
+ role: 'developer',
93
+ projectPath: '/tmp',
94
+ maxSteps: MAX_STEPS,
95
+ model: { provider: 'deepseek', modelId: 'deepseek-chat' },
96
+ } as unknown as CrewlyAgentConfig;
97
+
98
+ /** Drive the private recovery loop with a scripted sequence of attempts. */
99
+ async function runLoop(): Promise<AgentRunResult> {
100
+ return await (runner as unknown as {
101
+ executeRunWithStreamText(tools: Record<string, unknown>, signal: AbortSignal): Promise<AgentRunResult>;
102
+ }).executeRunWithStreamText({}, new AbortController().signal);
103
+ }
104
+
105
+ beforeEach(() => {
106
+ runner = new AgentRunnerService(config);
107
+ attempts = [];
108
+ attemptSpy = vi.fn(async () => attempts.shift() ?? runResult());
109
+ (runner as unknown as Record<string, unknown>).attemptWithErrorRetries = attemptSpy;
110
+ });
111
+
112
+ it('returns a healthy turn untouched, with no extra model call', async () => {
113
+ attempts = [runResult({ text: 'done', finishReason: 'stop' })];
114
+ const out = await runLoop();
115
+ expect(out.text).toBe('done');
116
+ expect(out.incomplete).toBeUndefined();
117
+ expect(attemptSpy).toHaveBeenCalledTimes(1);
118
+ });
119
+
120
+ it('continues a truncated turn until the model stops, and returns the joined answer', async () => {
121
+ attempts = [
122
+ runResult({ text: 'first half', finishReason: 'length' }),
123
+ runResult({ text: 'second half', finishReason: 'stop' }),
124
+ ];
125
+ const out = await runLoop();
126
+ expect(out.text).toBe('first half\n\nsecond half');
127
+ expect(out.incomplete).toBeUndefined();
128
+ expect(attemptSpy).toHaveBeenCalledTimes(2);
129
+ });
130
+
131
+ it('gives up after the continuation budget and reports what it kept', async () => {
132
+ attempts = Array.from({ length: 10 }, (_, i) => runResult({ text: `chunk ${i}`, finishReason: 'length' }));
133
+ const out = await runLoop();
134
+ expect(attemptSpy).toHaveBeenCalledTimes(CREWLY_AGENT_DEFAULTS.MAX_CONTINUATIONS + 1);
135
+ expect(out.incomplete).toMatchObject({ reason: 'truncated', recoveryAttempts: CREWLY_AGENT_DEFAULTS.MAX_CONTINUATIONS });
136
+ expect(out.text).toContain('chunk 0');
137
+ });
138
+
139
+ it('recovers the deepseek case: an abnormal finish retried once that then completes', async () => {
140
+ attempts = [
141
+ runResult({ text: "I'll create the team", finishReason: 'other' }),
142
+ runResult({ text: 'Team created.', finishReason: 'stop' }),
143
+ ];
144
+ const out = await runLoop();
145
+ expect(out.incomplete).toBeUndefined();
146
+ expect(out.text).toBe("I'll create the team\n\nTeam created.");
147
+ });
148
+
149
+ it('marks the turn incomplete when the provider keeps bailing', async () => {
150
+ attempts = [
151
+ runResult({ text: "I'll create the team", finishReason: 'other' }),
152
+ runResult({ text: '', finishReason: 'other' }),
153
+ ];
154
+ const out = await runLoop();
155
+ expect(attemptSpy).toHaveBeenCalledTimes(2);
156
+ expect(out.incomplete).toMatchObject({ reason: 'abnormal-finish', finishReason: 'other', recoveryAttempts: 1 });
157
+ });
158
+
159
+ it('does not retry a step-exhausted or content-filtered turn', async () => {
160
+ attempts = [runResult({ steps: MAX_STEPS, finishReason: 'tool-calls' })];
161
+ expect((await runLoop()).incomplete).toMatchObject({ reason: 'steps-exhausted', recoveryAttempts: 0 });
162
+ expect(attemptSpy).toHaveBeenCalledTimes(1);
163
+
164
+ attemptSpy.mockClear();
165
+ attempts = [runResult({ finishReason: 'content-filter' })];
166
+ expect((await runLoop()).incomplete).toMatchObject({ reason: 'content-filter' });
167
+ expect(attemptSpy).toHaveBeenCalledTimes(1);
168
+ });
169
+
170
+ it('pushes a continuation instruction into the conversation before retrying', async () => {
171
+ attempts = [runResult({ finishReason: 'length' }), runResult({ finishReason: 'stop' })];
172
+ await runLoop();
173
+ const messages = (runner as unknown as { state: { messages: Array<{ role: string; content: string }> } }).state.messages;
174
+ const nudge = messages.filter((m) => m.role === 'user').pop();
175
+ expect(nudge?.content).toMatch(/continue from exactly where you stopped/i);
176
+ });
177
+ });
@@ -203,6 +203,34 @@ export interface AgentRunResult {
203
203
  * See P0-1 design-review gate for I2.5 reasoning-pipe routing.
204
204
  */
205
205
  reasoning?: string | null;
206
+ /**
207
+ * Set when the turn did NOT run to a natural end: the model was cut off,
208
+ * the provider bailed, or the step budget ran out. The text is then a
209
+ * fragment of an answer, not an answer.
210
+ */
211
+ incomplete?: IncompleteRun;
212
+ }
213
+
214
+ /** Why a turn ended early. */
215
+ export type IncompleteReason =
216
+ /** Ran out of output tokens and continuation attempts. */
217
+ | 'truncated'
218
+ /** Provider returned an unmapped/error finish (`other`, `error`, `unknown`). */
219
+ | 'abnormal-finish'
220
+ /** Hit `maxSteps` with work still outstanding. */
221
+ | 'steps-exhausted'
222
+ /** Provider refused on content grounds. */
223
+ | 'content-filter';
224
+
225
+ /** Describes a turn that ended early. */
226
+ export interface IncompleteRun {
227
+ reason: IncompleteReason;
228
+ /** One line a human can act on. */
229
+ detail: string;
230
+ /** Raw provider finish reason of the final attempt. */
231
+ finishReason: string;
232
+ /** Recovery attempts made before giving up. */
233
+ recoveryAttempts: number;
206
234
  }
207
235
 
208
236
  /**
@@ -524,6 +552,10 @@ export interface SecurityGuardrailConfig {
524
552
  export const CREWLY_AGENT_DEFAULTS = {
525
553
  /** Default max reasoning steps per generateText call (high to mimic unlimited like Claude Code) */
526
554
  MAX_STEPS: 500,
555
+ /** Continuations allowed for a turn cut off by the output-token limit. */
556
+ MAX_CONTINUATIONS: 3,
557
+ /** Re-runs allowed after the provider ended a turn abnormally. */
558
+ MAX_ABNORMAL_RETRIES: 1,
527
559
  /** Maximum tool calls allowed per single response to prevent polling dead-loops */
528
560
  MAX_TOOL_CALLS_PER_RESPONSE: 15,
529
561
  /** Consecutive identical tool calls before aborting (loop detection) */