crewly 1.12.0 → 1.13.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/backend/backend/src/controllers/chat/chat.controller.d.ts.map +1 -1
- package/dist/backend/backend/src/controllers/chat/chat.controller.js +29 -7
- package/dist/backend/backend/src/controllers/chat/chat.controller.js.map +1 -1
- package/dist/backend/backend/src/controllers/cloud/cloud.controller.d.ts.map +1 -1
- package/dist/backend/backend/src/controllers/cloud/cloud.controller.js +1 -1
- package/dist/backend/backend/src/controllers/cloud/cloud.controller.js.map +1 -1
- package/dist/backend/backend/src/services/agent/agent-registration.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/agent/agent-registration.service.js +8 -1
- package/dist/backend/backend/src/services/agent/agent-registration.service.js.map +1 -1
- package/dist/backend/backend/src/services/agent/crewly-agent/crewly-agent-external-runtime.service.d.ts +78 -2
- package/dist/backend/backend/src/services/agent/crewly-agent/crewly-agent-external-runtime.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/agent/crewly-agent/crewly-agent-external-runtime.service.js +240 -44
- package/dist/backend/backend/src/services/agent/crewly-agent/crewly-agent-external-runtime.service.js.map +1 -1
- package/dist/backend/backend/src/services/agent/crewly-agent/types.d.ts +9 -0
- package/dist/backend/backend/src/services/agent/crewly-agent/types.d.ts.map +1 -1
- package/dist/backend/backend/src/services/agent/crewly-agent/types.js +9 -0
- package/dist/backend/backend/src/services/agent/crewly-agent/types.js.map +1 -1
- package/dist/backend/backend/src/services/cloud/cloud-client.service.d.ts +5 -0
- package/dist/backend/backend/src/services/cloud/cloud-client.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/cloud/cloud-client.service.js +7 -0
- package/dist/backend/backend/src/services/cloud/cloud-client.service.js.map +1 -1
- package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.js +3 -0
- package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.js.map +1 -1
- package/dist/backend/backend/src/services/orchestrator/commitment-approval-guard.d.ts.map +1 -1
- package/dist/backend/backend/src/services/orchestrator/commitment-approval-guard.js +26 -0
- package/dist/backend/backend/src/services/orchestrator/commitment-approval-guard.js.map +1 -1
- package/dist/backend/backend/src/services/orchestrator/onboarding/materialize-team.d.ts.map +1 -1
- package/dist/backend/backend/src/services/orchestrator/onboarding/materialize-team.js +19 -3
- package/dist/backend/backend/src/services/orchestrator/onboarding/materialize-team.js.map +1 -1
- package/dist/backend/backend/src/services/wiki/wiki-migrate.service.d.ts +3 -0
- package/dist/backend/backend/src/services/wiki/wiki-migrate.service.d.ts.map +1 -1
- package/dist/backend/backend/src/services/wiki/wiki-migrate.service.js +21 -10
- package/dist/backend/backend/src/services/wiki/wiki-migrate.service.js.map +1 -1
- package/dist/backend/backend/src/websocket/chat.gateway.d.ts.map +1 -1
- package/dist/backend/backend/src/websocket/chat.gateway.js +7 -1
- package/dist/backend/backend/src/websocket/chat.gateway.js.map +1 -1
- package/dist/cli/backend/src/services/cloud/cloud-client.service.d.ts +5 -0
- package/dist/cli/backend/src/services/cloud/cloud-client.service.d.ts.map +1 -1
- package/dist/cli/backend/src/services/cloud/cloud-client.service.js +7 -0
- package/dist/cli/backend/src/services/cloud/cloud-client.service.js.map +1 -1
- package/frontend/dist/assets/index-aa737fd3.js +5254 -0
- package/frontend/dist/index.html +1 -1
- package/package.json +1 -1
- package/packages/crewly-agent/src/cli.ts +16 -6
- package/packages/crewly-agent/src/runtime/agent-runner.service.test.ts +137 -3
- package/packages/crewly-agent/src/runtime/agent-runner.service.ts +100 -6
- package/packages/crewly-agent/src/runtime/tool-registry.test.ts +19 -0
- package/packages/crewly-agent/src/runtime/tool-registry.ts +4 -1
- package/packages/crewly-agent/src/runtime/types.ts +9 -0
- package/frontend/dist/assets/index-890d3f9d.js +0 -4961
package/frontend/dist/index.html
CHANGED
|
@@ -7,7 +7,7 @@
|
|
|
7
7
|
<meta name="color-scheme" content="dark" />
|
|
8
8
|
<!-- Nunito font is self-hosted via @fontsource/nunito (imported in main.tsx) -->
|
|
9
9
|
<title>Crewly AI Studio</title>
|
|
10
|
-
<script type="module" crossorigin src="/assets/index-
|
|
10
|
+
<script type="module" crossorigin src="/assets/index-aa737fd3.js"></script>
|
|
11
11
|
<link rel="stylesheet" href="/assets/index-8205ea5e.css">
|
|
12
12
|
</head>
|
|
13
13
|
<body class="bg-background-dark font-display text-text-primary-dark">
|
package/package.json
CHANGED
|
@@ -10,15 +10,19 @@ import type {
|
|
|
10
10
|
|
|
11
11
|
type ParentMessage =
|
|
12
12
|
| { type: 'init'; config: CrewlyAgentConfig }
|
|
13
|
-
| { type: 'run'; message: string; conversationId?: string; metadata?: Record<string, string> }
|
|
13
|
+
| { type: 'run'; runId?: string; message: string; conversationId?: string; metadata?: Record<string, string> }
|
|
14
14
|
| { type: 'abort' }
|
|
15
15
|
| { type: 'get-state' }
|
|
16
16
|
| { type: 'shutdown' };
|
|
17
17
|
|
|
18
18
|
type WorkerMessage =
|
|
19
19
|
| { type: 'ready' }
|
|
20
|
-
|
|
21
|
-
|
|
20
|
+
// `runId` is echoed straight back from the `run` that produced it so the
|
|
21
|
+
// parent can match a reply to its request. Without it the parent has to
|
|
22
|
+
// guess, and guessing wrong books one conversation's answer against
|
|
23
|
+
// another. Optional so an older parent that sends no id still works.
|
|
24
|
+
| { type: 'result'; runId?: string; data: AgentRunResult }
|
|
25
|
+
| { type: 'error'; runId?: string; error: string; code?: string }
|
|
22
26
|
| { type: 'log'; level: 'debug' | 'info' | 'warn' | 'error'; message: string }
|
|
23
27
|
| { type: 'stream'; event: 'text'; data: { chunk: string } }
|
|
24
28
|
| { type: 'stream'; event: 'toolStart'; data: { toolName: string; args: Record<string, unknown> } }
|
|
@@ -71,9 +75,10 @@ async function handleMessage(message: ParentMessage): Promise<void> {
|
|
|
71
75
|
runner = null;
|
|
72
76
|
}
|
|
73
77
|
break;
|
|
74
|
-
case 'run':
|
|
78
|
+
case 'run': {
|
|
79
|
+
const { runId } = message;
|
|
75
80
|
if (!runner || !runner.isInitialized()) {
|
|
76
|
-
send({ type: 'error', error: 'Worker not initialized', code: 'NOT_INITIALIZED' });
|
|
81
|
+
send({ type: 'error', runId, error: 'Worker not initialized', code: 'NOT_INITIALIZED' });
|
|
77
82
|
return;
|
|
78
83
|
}
|
|
79
84
|
currentAbort = new AbortController();
|
|
@@ -88,16 +93,18 @@ async function handleMessage(message: ParentMessage): Promise<void> {
|
|
|
88
93
|
},
|
|
89
94
|
);
|
|
90
95
|
currentAbort = null;
|
|
91
|
-
send({ type: 'result', data: result });
|
|
96
|
+
send({ type: 'result', runId, data: result });
|
|
92
97
|
} catch (error) {
|
|
93
98
|
currentAbort = null;
|
|
94
99
|
send({
|
|
95
100
|
type: 'error',
|
|
101
|
+
runId,
|
|
96
102
|
error: error instanceof Error ? error.message : String(error),
|
|
97
103
|
code: 'RUN_FAILED',
|
|
98
104
|
});
|
|
99
105
|
}
|
|
100
106
|
break;
|
|
107
|
+
}
|
|
101
108
|
case 'abort':
|
|
102
109
|
if (currentAbort) {
|
|
103
110
|
currentAbort.abort();
|
|
@@ -145,6 +152,9 @@ rl.on('line', (line) => {
|
|
|
145
152
|
handleMessage(message).catch((error) => {
|
|
146
153
|
send({
|
|
147
154
|
type: 'error',
|
|
155
|
+
// Carry the correlation id here too — an unhandled failure must settle
|
|
156
|
+
// the run it belongs to, not whichever run happens to be oldest.
|
|
157
|
+
runId: message.type === 'run' ? message.runId : undefined,
|
|
148
158
|
error: error instanceof Error ? error.message : String(error),
|
|
149
159
|
code: 'UNHANDLED',
|
|
150
160
|
});
|
|
@@ -1,8 +1,8 @@
|
|
|
1
|
-
import { describe, it, expect, beforeEach, vi, type Mocked, type MockInstance } from 'vitest';
|
|
2
|
-
import { AgentRunnerService, ToolCallLoopDetector } from './agent-runner.service.js';
|
|
1
|
+
import { describe, it, expect, beforeEach, afterEach, vi, type Mocked, type MockInstance } from 'vitest';
|
|
2
|
+
import { AgentRunnerService, AgentRunTimeoutError, ToolCallLoopDetector } from './agent-runner.service.js';
|
|
3
3
|
import { ModelManager } from './model-manager.js';
|
|
4
4
|
import { CrewlyApiClient } from './api-client.js';
|
|
5
|
-
import type
|
|
5
|
+
import { CREWLY_AGENT_DEFAULTS, type CrewlyAgentConfig, type SecurityPolicy, type AuditEntry } from './types.js';
|
|
6
6
|
|
|
7
7
|
describe('AgentRunnerService', () => {
|
|
8
8
|
let runner: AgentRunnerService;
|
|
@@ -2352,4 +2352,138 @@ describe('B4 — DeepSeek tool_choice passthrough regression', () => {
|
|
|
2352
2352
|
expect(runner.getConversationCount()).toBe(1);
|
|
2353
2353
|
});
|
|
2354
2354
|
});
|
|
2355
|
+
|
|
2356
|
+
});
|
|
2357
|
+
|
|
2358
|
+
/**
|
|
2359
|
+
* Wall-clock run budget — the 假死 (silent-catatonia) guard.
|
|
2360
|
+
*
|
|
2361
|
+
* Regression cover for the production incident where a stalled model
|
|
2362
|
+
* connection hung `await streamResult` forever. `MESSAGE_TIMEOUT_MS` and
|
|
2363
|
+
* `MODEL_TIMEOUT_MS` existed as documented constants but nothing ever read
|
|
2364
|
+
* them, so the orchestrator went silent for 12 days without a single log
|
|
2365
|
+
* line. These tests assert the budget is actually ENFORCED, not merely
|
|
2366
|
+
* declared.
|
|
2367
|
+
*/
|
|
2368
|
+
describe('AgentRunnerService run wall-clock budget', () => {
|
|
2369
|
+
let runner: AgentRunnerService;
|
|
2370
|
+
let mockGenerateText: vi.Mock<any>;
|
|
2371
|
+
let mockModelManager: any;
|
|
2372
|
+
let mockApiClient: any;
|
|
2373
|
+
|
|
2374
|
+
const baseConfig: CrewlyAgentConfig = {
|
|
2375
|
+
model: { provider: 'anthropic', modelId: 'claude-sonnet-4-20250514', temperature: 0.3, maxTokens: 8192 },
|
|
2376
|
+
maxSteps: 10,
|
|
2377
|
+
sessionName: 'budget-session',
|
|
2378
|
+
apiBaseUrl: 'http://localhost:8787',
|
|
2379
|
+
systemPrompt: 'You are a test agent.',
|
|
2380
|
+
maxHistoryMessages: 20,
|
|
2381
|
+
compactionThreshold: 0.8,
|
|
2382
|
+
};
|
|
2383
|
+
|
|
2384
|
+
beforeEach(async () => {
|
|
2385
|
+
vi.clearAllMocks();
|
|
2386
|
+
vi.useFakeTimers();
|
|
2387
|
+
|
|
2388
|
+
mockGenerateText = vi.fn<any>();
|
|
2389
|
+
mockModelManager = {
|
|
2390
|
+
getModel: vi.fn<any>().mockResolvedValue({ provider: 'mock', modelId: 'test-model' }),
|
|
2391
|
+
getAvailableProviders: vi.fn<any>(),
|
|
2392
|
+
clearCache: vi.fn<any>(),
|
|
2393
|
+
consumeDeepseekReasoning: vi.fn<any>().mockResolvedValue(null),
|
|
2394
|
+
};
|
|
2395
|
+
mockApiClient = { get: vi.fn<any>(), post: vi.fn<any>(), delete: vi.fn<any>() };
|
|
2396
|
+
|
|
2397
|
+
runner = new AgentRunnerService(baseConfig, mockModelManager, mockApiClient);
|
|
2398
|
+
runner._generateTextFn = mockGenerateText;
|
|
2399
|
+
await runner.initialize();
|
|
2400
|
+
});
|
|
2401
|
+
|
|
2402
|
+
afterEach(() => {
|
|
2403
|
+
vi.useRealTimers();
|
|
2404
|
+
});
|
|
2405
|
+
|
|
2406
|
+
it('rejects with AgentRunTimeoutError when the model never responds', async () => {
|
|
2407
|
+
// A model call that never settles — exactly the stalled-socket shape.
|
|
2408
|
+
mockGenerateText.mockImplementation(() => new Promise(() => {}));
|
|
2409
|
+
|
|
2410
|
+
const runPromise = runner.run('will stall');
|
|
2411
|
+
const assertion = expect(runPromise).rejects.toBeInstanceOf(AgentRunTimeoutError);
|
|
2412
|
+
|
|
2413
|
+
await vi.advanceTimersByTimeAsync(CREWLY_AGENT_DEFAULTS.MESSAGE_TIMEOUT_MS + 1);
|
|
2414
|
+
|
|
2415
|
+
await assertion;
|
|
2416
|
+
});
|
|
2417
|
+
|
|
2418
|
+
it('does not wedge the queue — later messages still process after a timeout', async () => {
|
|
2419
|
+
// THE regression: a hung run left `processing` true forever, so every
|
|
2420
|
+
// subsequent message queued behind it and the agent went catatonic.
|
|
2421
|
+
mockGenerateText.mockImplementationOnce(() => new Promise(() => {}));
|
|
2422
|
+
|
|
2423
|
+
const stalled = runner.run('will stall');
|
|
2424
|
+
const stalledAssertion = expect(stalled).rejects.toBeInstanceOf(AgentRunTimeoutError);
|
|
2425
|
+
await vi.advanceTimersByTimeAsync(CREWLY_AGENT_DEFAULTS.MESSAGE_TIMEOUT_MS + 1);
|
|
2426
|
+
await stalledAssertion;
|
|
2427
|
+
|
|
2428
|
+
expect(runner.isProcessing()).toBe(false);
|
|
2429
|
+
|
|
2430
|
+
mockGenerateText.mockResolvedValueOnce({
|
|
2431
|
+
text: 'recovered',
|
|
2432
|
+
toolCalls: [],
|
|
2433
|
+
steps: [],
|
|
2434
|
+
usage: { inputTokens: 1, outputTokens: 1 },
|
|
2435
|
+
finishReason: 'stop',
|
|
2436
|
+
});
|
|
2437
|
+
|
|
2438
|
+
await expect(runner.run('after the stall')).resolves.toMatchObject({ text: 'recovered' });
|
|
2439
|
+
});
|
|
2440
|
+
|
|
2441
|
+
it('does not retry a budget expiry (it spans the retries already)', async () => {
|
|
2442
|
+
mockGenerateText.mockImplementation(() => new Promise(() => {}));
|
|
2443
|
+
|
|
2444
|
+
const runPromise = runner.run('will stall');
|
|
2445
|
+
const assertion = expect(runPromise).rejects.toBeInstanceOf(AgentRunTimeoutError);
|
|
2446
|
+
await vi.advanceTimersByTimeAsync(CREWLY_AGENT_DEFAULTS.MESSAGE_TIMEOUT_MS + 1);
|
|
2447
|
+
await assertion;
|
|
2448
|
+
|
|
2449
|
+
// One attempt only — a retried timeout would multiply the stall.
|
|
2450
|
+
expect(mockGenerateText).toHaveBeenCalledTimes(1);
|
|
2451
|
+
});
|
|
2452
|
+
|
|
2453
|
+
it('honors the per-model budget override for reasoning models', async () => {
|
|
2454
|
+
const reasoningRunner = new AgentRunnerService(
|
|
2455
|
+
{ ...baseConfig, model: { ...baseConfig.model, modelId: 'deepseek-reasoner' } },
|
|
2456
|
+
mockModelManager,
|
|
2457
|
+
mockApiClient,
|
|
2458
|
+
);
|
|
2459
|
+
reasoningRunner._generateTextFn = mockGenerateText;
|
|
2460
|
+
await reasoningRunner.initialize();
|
|
2461
|
+
mockGenerateText.mockImplementation(() => new Promise(() => {}));
|
|
2462
|
+
|
|
2463
|
+
const runPromise = reasoningRunner.run('slow chain-of-thought');
|
|
2464
|
+
const settled = vi.fn();
|
|
2465
|
+
runPromise.then(settled, settled);
|
|
2466
|
+
|
|
2467
|
+
// Still alive at the default budget — the override must win.
|
|
2468
|
+
await vi.advanceTimersByTimeAsync(CREWLY_AGENT_DEFAULTS.MESSAGE_TIMEOUT_MS + 1);
|
|
2469
|
+
expect(settled).not.toHaveBeenCalled();
|
|
2470
|
+
|
|
2471
|
+
const assertion = expect(runPromise).rejects.toBeInstanceOf(AgentRunTimeoutError);
|
|
2472
|
+
await vi.advanceTimersByTimeAsync(
|
|
2473
|
+
CREWLY_AGENT_DEFAULTS.MODEL_TIMEOUT_MS['deepseek-reasoner'],
|
|
2474
|
+
);
|
|
2475
|
+
await assertion;
|
|
2476
|
+
});
|
|
2477
|
+
|
|
2478
|
+
it('leaves a run that completes in time untouched', async () => {
|
|
2479
|
+
mockGenerateText.mockResolvedValueOnce({
|
|
2480
|
+
text: 'fast enough',
|
|
2481
|
+
toolCalls: [],
|
|
2482
|
+
steps: [],
|
|
2483
|
+
usage: { inputTokens: 1, outputTokens: 1 },
|
|
2484
|
+
finishReason: 'stop',
|
|
2485
|
+
});
|
|
2486
|
+
|
|
2487
|
+
await expect(runner.run('quick')).resolves.toMatchObject({ text: 'fast enough' });
|
|
2488
|
+
});
|
|
2355
2489
|
});
|
|
@@ -158,7 +158,10 @@ export class ToolCallLoopDetector {
|
|
|
158
158
|
}
|
|
159
159
|
if (this.consecutiveErrors >= this.errorThreshold) {
|
|
160
160
|
this.loopDetected = true;
|
|
161
|
-
|
|
161
|
+
// Serialize objects properly — `String(obj)` yields "[object Object]"
|
|
162
|
+
// and loses the actual error. Mirror isErrorResult()'s serialization.
|
|
163
|
+
const resultStr = typeof result === 'string' ? result : JSON.stringify(result);
|
|
164
|
+
this.loopReason = `Tool "${toolName}" returned errors ${this.consecutiveErrors} consecutive times. Last result: ${resultStr.slice(0, 200)}`;
|
|
162
165
|
return true;
|
|
163
166
|
}
|
|
164
167
|
} else {
|
|
@@ -198,6 +201,40 @@ export class ToolCallLoopDetector {
|
|
|
198
201
|
* const result = await runner.run('Check all team statuses');
|
|
199
202
|
* ```
|
|
200
203
|
*/
|
|
204
|
+
/**
|
|
205
|
+
* Thrown when a single agent run exceeds its wall-clock budget.
|
|
206
|
+
*
|
|
207
|
+
* Distinct class (rather than a message-sniffed "timeout" string) because
|
|
208
|
+
* {@link AgentRunnerService.isRecoverableError} treats anything mentioning
|
|
209
|
+
* "timeout" as retryable. A hard-budget expiry is deliberately NOT retryable:
|
|
210
|
+
* the budget covers the whole run including retries, so re-entering the retry
|
|
211
|
+
* loop would multiply the very stall the budget exists to bound.
|
|
212
|
+
*
|
|
213
|
+
* @example
|
|
214
|
+
* ```typescript
|
|
215
|
+
* try { await runner.run(msg); }
|
|
216
|
+
* catch (e) { if (e instanceof AgentRunTimeoutError) recycleAgent(); }
|
|
217
|
+
* ```
|
|
218
|
+
*/
|
|
219
|
+
export class AgentRunTimeoutError extends Error {
|
|
220
|
+
/** Wall-clock budget that was exceeded, in milliseconds. */
|
|
221
|
+
readonly timeoutMs: number;
|
|
222
|
+
|
|
223
|
+
/**
|
|
224
|
+
* @param timeoutMs - The budget that was exceeded, in milliseconds
|
|
225
|
+
* @param modelId - Model the run was using, for operator diagnosis
|
|
226
|
+
*/
|
|
227
|
+
constructor(timeoutMs: number, modelId: string) {
|
|
228
|
+
super(
|
|
229
|
+
`Agent run exceeded its ${timeoutMs}ms wall-clock budget (model: ${modelId}) `
|
|
230
|
+
+ `and was hard-aborted. The model connection stalled without delivering a `
|
|
231
|
+
+ `response — see CREWLY_AGENT_MESSAGE_TIMEOUT_MS to tune the budget.`,
|
|
232
|
+
);
|
|
233
|
+
this.name = 'AgentRunTimeoutError';
|
|
234
|
+
this.timeoutMs = timeoutMs;
|
|
235
|
+
}
|
|
236
|
+
}
|
|
237
|
+
|
|
201
238
|
/** Function type for generateText — used for dependency injection in tests */
|
|
202
239
|
type GenerateTextFn = (opts: Record<string, unknown>) => Promise<Record<string, unknown>>;
|
|
203
240
|
|
|
@@ -824,19 +861,73 @@ export class AgentRunnerService {
|
|
|
824
861
|
externalAbortSignal.addEventListener('abort', () => runAbort.abort(), { once: true });
|
|
825
862
|
}
|
|
826
863
|
|
|
864
|
+
// Wall-clock budget for the ENTIRE run (all retries included).
|
|
865
|
+
//
|
|
866
|
+
// Without this, a model connection that opens but then goes silent — no
|
|
867
|
+
// bytes, no FIN, no error — hangs `await streamResult` forever: undici
|
|
868
|
+
// applies no idle deadline to a streaming body. That stalled promise wedges
|
|
869
|
+
// processQueue permanently (`this.processing` stays true), so every later
|
|
870
|
+
// message queues behind it and the agent goes silently catatonic: the
|
|
871
|
+
// process is alive and heartbeating, but no run ever completes and no error
|
|
872
|
+
// is ever raised. Bounding the run converts that invisible hang into a
|
|
873
|
+
// normal rejection the caller can log, surface, and recover from.
|
|
874
|
+
const timeoutMs = this.resolveRunTimeoutMs();
|
|
875
|
+
let hardTimer: ReturnType<typeof setTimeout> | undefined;
|
|
876
|
+
let softTimer: ReturnType<typeof setTimeout> | undefined;
|
|
877
|
+
|
|
827
878
|
try {
|
|
828
879
|
// If a test override is set, use generateText path (backward compatible)
|
|
829
|
-
|
|
830
|
-
|
|
831
|
-
|
|
880
|
+
const runPromise = this._generateTextFn
|
|
881
|
+
? this.executeRunWithGenerateText(tools, runAbort.signal)
|
|
882
|
+
// Production path: streamText for real-time feedback
|
|
883
|
+
: this.executeRunWithStreamText(tools, runAbort.signal);
|
|
884
|
+
|
|
885
|
+
// The loser of the race stays pending. Attach a no-op catch so its
|
|
886
|
+
// eventual abort-rejection is never an unhandled rejection (which
|
|
887
|
+
// crashes the agent process under Node's default policy).
|
|
888
|
+
runPromise.catch(() => { /* settled by the race below */ });
|
|
889
|
+
|
|
890
|
+
const timeoutPromise = new Promise<never>((_, reject) => {
|
|
891
|
+
softTimer = setTimeout(() => {
|
|
892
|
+
console.warn(
|
|
893
|
+
`[AgentRunner] Run exceeded soft warning threshold `
|
|
894
|
+
+ `(${CREWLY_AGENT_DEFAULTS.MESSAGE_SOFT_WARNING_MS}ms) — still waiting on the model.`,
|
|
895
|
+
);
|
|
896
|
+
}, CREWLY_AGENT_DEFAULTS.MESSAGE_SOFT_WARNING_MS);
|
|
897
|
+
|
|
898
|
+
hardTimer = setTimeout(() => {
|
|
899
|
+
// Abort first so the in-flight HTTP request and any tool work are
|
|
900
|
+
// torn down, then reject — a well-behaved provider unwinds on the
|
|
901
|
+
// signal, and a wedged one no longer holds the queue hostage.
|
|
902
|
+
runAbort.abort();
|
|
903
|
+
reject(new AgentRunTimeoutError(timeoutMs, this.config.model.modelId));
|
|
904
|
+
}, timeoutMs);
|
|
905
|
+
});
|
|
832
906
|
|
|
833
|
-
|
|
834
|
-
return await this.executeRunWithStreamText(tools, runAbort.signal);
|
|
907
|
+
return await Promise.race([runPromise, timeoutPromise]);
|
|
835
908
|
} finally {
|
|
909
|
+
if (hardTimer) clearTimeout(hardTimer);
|
|
910
|
+
if (softTimer) clearTimeout(softTimer);
|
|
836
911
|
this.currentRunAbort = null;
|
|
837
912
|
}
|
|
838
913
|
}
|
|
839
914
|
|
|
915
|
+
/**
|
|
916
|
+
* Resolve the wall-clock budget for one run.
|
|
917
|
+
*
|
|
918
|
+
* Per-model overrides win over the global default: reasoning models emit
|
|
919
|
+
* chain-of-thought tokens far slower than plain content, so a budget tuned
|
|
920
|
+
* for a chat model would kill them mid-thought.
|
|
921
|
+
*
|
|
922
|
+
* @returns Budget in milliseconds for the currently configured model
|
|
923
|
+
*/
|
|
924
|
+
private resolveRunTimeoutMs(): number {
|
|
925
|
+
return (
|
|
926
|
+
CREWLY_AGENT_DEFAULTS.MODEL_TIMEOUT_MS[this.config.model.modelId]
|
|
927
|
+
?? CREWLY_AGENT_DEFAULTS.MESSAGE_TIMEOUT_MS
|
|
928
|
+
);
|
|
929
|
+
}
|
|
930
|
+
|
|
840
931
|
/**
|
|
841
932
|
* Check if an error is recoverable and eligible for automatic retry.
|
|
842
933
|
*
|
|
@@ -850,6 +941,9 @@ export class AgentRunnerService {
|
|
|
850
941
|
*/
|
|
851
942
|
private isRecoverableError(error: unknown): boolean {
|
|
852
943
|
if (!(error instanceof Error)) return false;
|
|
944
|
+
// Hard-budget expiry is terminal by construction: the budget spans the
|
|
945
|
+
// whole run, so retrying would just re-enter the stall it bounded.
|
|
946
|
+
if (error instanceof AgentRunTimeoutError) return false;
|
|
853
947
|
const msg = error.message.toLowerCase();
|
|
854
948
|
const statusMatch = msg.match(/\b(429|5\d{2})\b/);
|
|
855
949
|
if (statusMatch) return true;
|
|
@@ -2186,6 +2186,25 @@ describe('Tool Registry', () => {
|
|
|
2186
2186
|
expect(validateFn('rm -rf ./dist')).toBeNull();
|
|
2187
2187
|
expect(validateFn('rm -rf node_modules')).toBeNull();
|
|
2188
2188
|
});
|
|
2189
|
+
|
|
2190
|
+
it('should signal the block is permanent so the model stops retrying', () => {
|
|
2191
|
+
// A blocked command is a policy decision, not a transient failure. When
|
|
2192
|
+
// the message read like a retryable error the model retried variations
|
|
2193
|
+
// until the loop detector aborted the whole run, so the wording carries
|
|
2194
|
+
// real behavioural weight and is worth pinning.
|
|
2195
|
+
const reason = validateFn('kill -9 $$');
|
|
2196
|
+
|
|
2197
|
+
expect(reason).not.toBeNull();
|
|
2198
|
+
expect(reason).toMatch(/permanently blocked/i);
|
|
2199
|
+
expect(reason).toMatch(/NOT a transient error/i);
|
|
2200
|
+
expect(reason).toMatch(/do NOT retry/i);
|
|
2201
|
+
});
|
|
2202
|
+
|
|
2203
|
+
it('should still identify which pattern caused the block', () => {
|
|
2204
|
+
// The matched pattern stays in the message: without it the model cannot
|
|
2205
|
+
// tell which part of a compound command was rejected.
|
|
2206
|
+
expect(validateFn('shutdown -h now')).toMatch(/shutdown/);
|
|
2207
|
+
});
|
|
2189
2208
|
});
|
|
2190
2209
|
|
|
2191
2210
|
describe('bash_exec tool', () => {
|
|
@@ -79,7 +79,10 @@ export const APPROVAL_REQUIRED_BASH_PATTERNS: Array<{ pattern: RegExp; label: st
|
|
|
79
79
|
export function validateBashCommand(command: string): string | null {
|
|
80
80
|
for (const pattern of BLOCKED_COMMAND_PATTERNS) {
|
|
81
81
|
if (pattern.test(command)) {
|
|
82
|
-
|
|
82
|
+
// Phrased to signal the model this is a PERMANENT policy block, not a
|
|
83
|
+
// transient failure — otherwise it retries variations until the loop
|
|
84
|
+
// detector aborts the whole run.
|
|
85
|
+
return `Command permanently blocked by security policy (matched ${pattern}). This is NOT a transient error — do NOT retry this or a variation. Achieve the goal a different way (use a Crewly API/tool, or skip this step).`;
|
|
83
86
|
}
|
|
84
87
|
}
|
|
85
88
|
return null;
|
|
@@ -540,6 +540,15 @@ export const CREWLY_AGENT_DEFAULTS = {
|
|
|
540
540
|
/** Soft warning threshold in milliseconds — logs a warning but does not kill the request.
|
|
541
541
|
* Override via CREWLY_AGENT_MESSAGE_SOFT_WARNING_MS env var. */
|
|
542
542
|
MESSAGE_SOFT_WARNING_MS: Number(process.env.CREWLY_AGENT_MESSAGE_SOFT_WARNING_MS) || 240_000,
|
|
543
|
+
/** Grace period added on top of the agent's own run budget before the PARENT
|
|
544
|
+
* gives up waiting on the child's IPC reply.
|
|
545
|
+
*
|
|
546
|
+
* The child enforces the run budget itself and reports a clean error, so in
|
|
547
|
+
* normal operation the parent timer never fires. It exists for the case the
|
|
548
|
+
* child can no longer report at all — blocked event loop, broken stdout pipe,
|
|
549
|
+
* SIGSTOP. The grace keeps the parent from racing the child and mislabelling
|
|
550
|
+
* a normal in-budget timeout as a wedged runtime. */
|
|
551
|
+
IPC_RUN_TIMEOUT_GRACE_MS: 30_000,
|
|
543
552
|
/** Cooldown period in milliseconds after rate limit retries are exhausted */
|
|
544
553
|
RATE_LIMIT_COOLDOWN_MS: 300_000,
|
|
545
554
|
/** Maximum number of automatic retries for recoverable errors (429, 5xx, network) */
|