crewly 1.12.0 → 1.13.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. package/dist/backend/backend/src/controllers/chat/chat.controller.d.ts.map +1 -1
  2. package/dist/backend/backend/src/controllers/chat/chat.controller.js +29 -7
  3. package/dist/backend/backend/src/controllers/chat/chat.controller.js.map +1 -1
  4. package/dist/backend/backend/src/controllers/cloud/cloud.controller.d.ts.map +1 -1
  5. package/dist/backend/backend/src/controllers/cloud/cloud.controller.js +1 -1
  6. package/dist/backend/backend/src/controllers/cloud/cloud.controller.js.map +1 -1
  7. package/dist/backend/backend/src/services/agent/agent-registration.service.d.ts.map +1 -1
  8. package/dist/backend/backend/src/services/agent/agent-registration.service.js +8 -1
  9. package/dist/backend/backend/src/services/agent/agent-registration.service.js.map +1 -1
  10. package/dist/backend/backend/src/services/agent/crewly-agent/crewly-agent-external-runtime.service.d.ts +78 -2
  11. package/dist/backend/backend/src/services/agent/crewly-agent/crewly-agent-external-runtime.service.d.ts.map +1 -1
  12. package/dist/backend/backend/src/services/agent/crewly-agent/crewly-agent-external-runtime.service.js +240 -44
  13. package/dist/backend/backend/src/services/agent/crewly-agent/crewly-agent-external-runtime.service.js.map +1 -1
  14. package/dist/backend/backend/src/services/agent/crewly-agent/types.d.ts +9 -0
  15. package/dist/backend/backend/src/services/agent/crewly-agent/types.d.ts.map +1 -1
  16. package/dist/backend/backend/src/services/agent/crewly-agent/types.js +9 -0
  17. package/dist/backend/backend/src/services/agent/crewly-agent/types.js.map +1 -1
  18. package/dist/backend/backend/src/services/cloud/cloud-client.service.d.ts +5 -0
  19. package/dist/backend/backend/src/services/cloud/cloud-client.service.d.ts.map +1 -1
  20. package/dist/backend/backend/src/services/cloud/cloud-client.service.js +7 -0
  21. package/dist/backend/backend/src/services/cloud/cloud-client.service.js.map +1 -1
  22. package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.d.ts.map +1 -1
  23. package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.js +3 -0
  24. package/dist/backend/backend/src/services/cloud/mobile-api-relay.service.js.map +1 -1
  25. package/dist/backend/backend/src/services/orchestrator/commitment-approval-guard.d.ts.map +1 -1
  26. package/dist/backend/backend/src/services/orchestrator/commitment-approval-guard.js +26 -0
  27. package/dist/backend/backend/src/services/orchestrator/commitment-approval-guard.js.map +1 -1
  28. package/dist/backend/backend/src/services/orchestrator/onboarding/materialize-team.d.ts.map +1 -1
  29. package/dist/backend/backend/src/services/orchestrator/onboarding/materialize-team.js +19 -3
  30. package/dist/backend/backend/src/services/orchestrator/onboarding/materialize-team.js.map +1 -1
  31. package/dist/backend/backend/src/services/wiki/wiki-migrate.service.d.ts +3 -0
  32. package/dist/backend/backend/src/services/wiki/wiki-migrate.service.d.ts.map +1 -1
  33. package/dist/backend/backend/src/services/wiki/wiki-migrate.service.js +21 -10
  34. package/dist/backend/backend/src/services/wiki/wiki-migrate.service.js.map +1 -1
  35. package/dist/backend/backend/src/websocket/chat.gateway.d.ts.map +1 -1
  36. package/dist/backend/backend/src/websocket/chat.gateway.js +7 -1
  37. package/dist/backend/backend/src/websocket/chat.gateway.js.map +1 -1
  38. package/dist/cli/backend/src/services/cloud/cloud-client.service.d.ts +5 -0
  39. package/dist/cli/backend/src/services/cloud/cloud-client.service.d.ts.map +1 -1
  40. package/dist/cli/backend/src/services/cloud/cloud-client.service.js +7 -0
  41. package/dist/cli/backend/src/services/cloud/cloud-client.service.js.map +1 -1
  42. package/frontend/dist/assets/index-aa737fd3.js +5254 -0
  43. package/frontend/dist/index.html +1 -1
  44. package/package.json +1 -1
  45. package/packages/crewly-agent/src/cli.ts +16 -6
  46. package/packages/crewly-agent/src/runtime/agent-runner.service.test.ts +137 -3
  47. package/packages/crewly-agent/src/runtime/agent-runner.service.ts +100 -6
  48. package/packages/crewly-agent/src/runtime/tool-registry.test.ts +19 -0
  49. package/packages/crewly-agent/src/runtime/tool-registry.ts +4 -1
  50. package/packages/crewly-agent/src/runtime/types.ts +9 -0
  51. package/frontend/dist/assets/index-890d3f9d.js +0 -4961
@@ -7,7 +7,7 @@
7
7
  <meta name="color-scheme" content="dark" />
8
8
  <!-- Nunito font is self-hosted via @fontsource/nunito (imported in main.tsx) -->
9
9
  <title>Crewly AI Studio</title>
10
- <script type="module" crossorigin src="/assets/index-890d3f9d.js"></script>
10
+ <script type="module" crossorigin src="/assets/index-aa737fd3.js"></script>
11
11
  <link rel="stylesheet" href="/assets/index-8205ea5e.css">
12
12
  </head>
13
13
  <body class="bg-background-dark font-display text-text-primary-dark">
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "crewly",
3
- "version": "1.12.0",
3
+ "version": "1.13.0",
4
4
  "type": "module",
5
5
  "description": "Multi-agent orchestration platform for AI coding teams — coordinates Claude Code, Gemini CLI, and Codex agents with a real-time web dashboard",
6
6
  "workspaces": [
@@ -10,15 +10,19 @@ import type {
10
10
 
11
11
  type ParentMessage =
12
12
  | { type: 'init'; config: CrewlyAgentConfig }
13
- | { type: 'run'; message: string; conversationId?: string; metadata?: Record<string, string> }
13
+ | { type: 'run'; runId?: string; message: string; conversationId?: string; metadata?: Record<string, string> }
14
14
  | { type: 'abort' }
15
15
  | { type: 'get-state' }
16
16
  | { type: 'shutdown' };
17
17
 
18
18
  type WorkerMessage =
19
19
  | { type: 'ready' }
20
- | { type: 'result'; data: AgentRunResult }
21
- | { type: 'error'; error: string; code?: string }
20
+ // `runId` is echoed straight back from the `run` that produced it so the
21
+ // parent can match a reply to its request. Without it the parent has to
22
+ // guess, and guessing wrong books one conversation's answer against
23
+ // another. Optional so an older parent that sends no id still works.
24
+ | { type: 'result'; runId?: string; data: AgentRunResult }
25
+ | { type: 'error'; runId?: string; error: string; code?: string }
22
26
  | { type: 'log'; level: 'debug' | 'info' | 'warn' | 'error'; message: string }
23
27
  | { type: 'stream'; event: 'text'; data: { chunk: string } }
24
28
  | { type: 'stream'; event: 'toolStart'; data: { toolName: string; args: Record<string, unknown> } }
@@ -71,9 +75,10 @@ async function handleMessage(message: ParentMessage): Promise<void> {
71
75
  runner = null;
72
76
  }
73
77
  break;
74
- case 'run':
78
+ case 'run': {
79
+ const { runId } = message;
75
80
  if (!runner || !runner.isInitialized()) {
76
- send({ type: 'error', error: 'Worker not initialized', code: 'NOT_INITIALIZED' });
81
+ send({ type: 'error', runId, error: 'Worker not initialized', code: 'NOT_INITIALIZED' });
77
82
  return;
78
83
  }
79
84
  currentAbort = new AbortController();
@@ -88,16 +93,18 @@ async function handleMessage(message: ParentMessage): Promise<void> {
88
93
  },
89
94
  );
90
95
  currentAbort = null;
91
- send({ type: 'result', data: result });
96
+ send({ type: 'result', runId, data: result });
92
97
  } catch (error) {
93
98
  currentAbort = null;
94
99
  send({
95
100
  type: 'error',
101
+ runId,
96
102
  error: error instanceof Error ? error.message : String(error),
97
103
  code: 'RUN_FAILED',
98
104
  });
99
105
  }
100
106
  break;
107
+ }
101
108
  case 'abort':
102
109
  if (currentAbort) {
103
110
  currentAbort.abort();
@@ -145,6 +152,9 @@ rl.on('line', (line) => {
145
152
  handleMessage(message).catch((error) => {
146
153
  send({
147
154
  type: 'error',
155
+ // Carry the correlation id here too — an unhandled failure must settle
156
+ // the run it belongs to, not whichever run happens to be oldest.
157
+ runId: message.type === 'run' ? message.runId : undefined,
148
158
  error: error instanceof Error ? error.message : String(error),
149
159
  code: 'UNHANDLED',
150
160
  });
@@ -1,8 +1,8 @@
1
- import { describe, it, expect, beforeEach, vi, type Mocked, type MockInstance } from 'vitest';
2
- import { AgentRunnerService, ToolCallLoopDetector } from './agent-runner.service.js';
1
+ import { describe, it, expect, beforeEach, afterEach, vi, type Mocked, type MockInstance } from 'vitest';
2
+ import { AgentRunnerService, AgentRunTimeoutError, ToolCallLoopDetector } from './agent-runner.service.js';
3
3
  import { ModelManager } from './model-manager.js';
4
4
  import { CrewlyApiClient } from './api-client.js';
5
- import type { CrewlyAgentConfig, SecurityPolicy, AuditEntry } from './types.js';
5
+ import { CREWLY_AGENT_DEFAULTS, type CrewlyAgentConfig, type SecurityPolicy, type AuditEntry } from './types.js';
6
6
 
7
7
  describe('AgentRunnerService', () => {
8
8
  let runner: AgentRunnerService;
@@ -2352,4 +2352,138 @@ describe('B4 — DeepSeek tool_choice passthrough regression', () => {
2352
2352
  expect(runner.getConversationCount()).toBe(1);
2353
2353
  });
2354
2354
  });
2355
+
2356
+ });
2357
+
2358
+ /**
2359
+ * Wall-clock run budget — the 假死 (silent-catatonia) guard.
2360
+ *
2361
+ * Regression cover for the production incident where a stalled model
2362
+ * connection hung `await streamResult` forever. `MESSAGE_TIMEOUT_MS` and
2363
+ * `MODEL_TIMEOUT_MS` existed as documented constants but nothing ever read
2364
+ * them, so the orchestrator went silent for 12 days without a single log
2365
+ * line. These tests assert the budget is actually ENFORCED, not merely
2366
+ * declared.
2367
+ */
2368
+ describe('AgentRunnerService run wall-clock budget', () => {
2369
+ let runner: AgentRunnerService;
2370
+ let mockGenerateText: vi.Mock<any>;
2371
+ let mockModelManager: any;
2372
+ let mockApiClient: any;
2373
+
2374
+ const baseConfig: CrewlyAgentConfig = {
2375
+ model: { provider: 'anthropic', modelId: 'claude-sonnet-4-20250514', temperature: 0.3, maxTokens: 8192 },
2376
+ maxSteps: 10,
2377
+ sessionName: 'budget-session',
2378
+ apiBaseUrl: 'http://localhost:8787',
2379
+ systemPrompt: 'You are a test agent.',
2380
+ maxHistoryMessages: 20,
2381
+ compactionThreshold: 0.8,
2382
+ };
2383
+
2384
+ beforeEach(async () => {
2385
+ vi.clearAllMocks();
2386
+ vi.useFakeTimers();
2387
+
2388
+ mockGenerateText = vi.fn<any>();
2389
+ mockModelManager = {
2390
+ getModel: vi.fn<any>().mockResolvedValue({ provider: 'mock', modelId: 'test-model' }),
2391
+ getAvailableProviders: vi.fn<any>(),
2392
+ clearCache: vi.fn<any>(),
2393
+ consumeDeepseekReasoning: vi.fn<any>().mockResolvedValue(null),
2394
+ };
2395
+ mockApiClient = { get: vi.fn<any>(), post: vi.fn<any>(), delete: vi.fn<any>() };
2396
+
2397
+ runner = new AgentRunnerService(baseConfig, mockModelManager, mockApiClient);
2398
+ runner._generateTextFn = mockGenerateText;
2399
+ await runner.initialize();
2400
+ });
2401
+
2402
+ afterEach(() => {
2403
+ vi.useRealTimers();
2404
+ });
2405
+
2406
+ it('rejects with AgentRunTimeoutError when the model never responds', async () => {
2407
+ // A model call that never settles — exactly the stalled-socket shape.
2408
+ mockGenerateText.mockImplementation(() => new Promise(() => {}));
2409
+
2410
+ const runPromise = runner.run('will stall');
2411
+ const assertion = expect(runPromise).rejects.toBeInstanceOf(AgentRunTimeoutError);
2412
+
2413
+ await vi.advanceTimersByTimeAsync(CREWLY_AGENT_DEFAULTS.MESSAGE_TIMEOUT_MS + 1);
2414
+
2415
+ await assertion;
2416
+ });
2417
+
2418
+ it('does not wedge the queue — later messages still process after a timeout', async () => {
2419
+ // THE regression: a hung run left `processing` true forever, so every
2420
+ // subsequent message queued behind it and the agent went catatonic.
2421
+ mockGenerateText.mockImplementationOnce(() => new Promise(() => {}));
2422
+
2423
+ const stalled = runner.run('will stall');
2424
+ const stalledAssertion = expect(stalled).rejects.toBeInstanceOf(AgentRunTimeoutError);
2425
+ await vi.advanceTimersByTimeAsync(CREWLY_AGENT_DEFAULTS.MESSAGE_TIMEOUT_MS + 1);
2426
+ await stalledAssertion;
2427
+
2428
+ expect(runner.isProcessing()).toBe(false);
2429
+
2430
+ mockGenerateText.mockResolvedValueOnce({
2431
+ text: 'recovered',
2432
+ toolCalls: [],
2433
+ steps: [],
2434
+ usage: { inputTokens: 1, outputTokens: 1 },
2435
+ finishReason: 'stop',
2436
+ });
2437
+
2438
+ await expect(runner.run('after the stall')).resolves.toMatchObject({ text: 'recovered' });
2439
+ });
2440
+
2441
+ it('does not retry a budget expiry (it spans the retries already)', async () => {
2442
+ mockGenerateText.mockImplementation(() => new Promise(() => {}));
2443
+
2444
+ const runPromise = runner.run('will stall');
2445
+ const assertion = expect(runPromise).rejects.toBeInstanceOf(AgentRunTimeoutError);
2446
+ await vi.advanceTimersByTimeAsync(CREWLY_AGENT_DEFAULTS.MESSAGE_TIMEOUT_MS + 1);
2447
+ await assertion;
2448
+
2449
+ // One attempt only — a retried timeout would multiply the stall.
2450
+ expect(mockGenerateText).toHaveBeenCalledTimes(1);
2451
+ });
2452
+
2453
+ it('honors the per-model budget override for reasoning models', async () => {
2454
+ const reasoningRunner = new AgentRunnerService(
2455
+ { ...baseConfig, model: { ...baseConfig.model, modelId: 'deepseek-reasoner' } },
2456
+ mockModelManager,
2457
+ mockApiClient,
2458
+ );
2459
+ reasoningRunner._generateTextFn = mockGenerateText;
2460
+ await reasoningRunner.initialize();
2461
+ mockGenerateText.mockImplementation(() => new Promise(() => {}));
2462
+
2463
+ const runPromise = reasoningRunner.run('slow chain-of-thought');
2464
+ const settled = vi.fn();
2465
+ runPromise.then(settled, settled);
2466
+
2467
+ // Still alive at the default budget — the override must win.
2468
+ await vi.advanceTimersByTimeAsync(CREWLY_AGENT_DEFAULTS.MESSAGE_TIMEOUT_MS + 1);
2469
+ expect(settled).not.toHaveBeenCalled();
2470
+
2471
+ const assertion = expect(runPromise).rejects.toBeInstanceOf(AgentRunTimeoutError);
2472
+ await vi.advanceTimersByTimeAsync(
2473
+ CREWLY_AGENT_DEFAULTS.MODEL_TIMEOUT_MS['deepseek-reasoner'],
2474
+ );
2475
+ await assertion;
2476
+ });
2477
+
2478
+ it('leaves a run that completes in time untouched', async () => {
2479
+ mockGenerateText.mockResolvedValueOnce({
2480
+ text: 'fast enough',
2481
+ toolCalls: [],
2482
+ steps: [],
2483
+ usage: { inputTokens: 1, outputTokens: 1 },
2484
+ finishReason: 'stop',
2485
+ });
2486
+
2487
+ await expect(runner.run('quick')).resolves.toMatchObject({ text: 'fast enough' });
2488
+ });
2355
2489
  });
@@ -158,7 +158,10 @@ export class ToolCallLoopDetector {
158
158
  }
159
159
  if (this.consecutiveErrors >= this.errorThreshold) {
160
160
  this.loopDetected = true;
161
- this.loopReason = `Tool "${toolName}" returned errors ${this.consecutiveErrors} consecutive times. Last result: ${String(result).slice(0, 200)}`;
161
+ // Serialize objects properly — `String(obj)` yields "[object Object]"
162
+ // and loses the actual error. Mirror isErrorResult()'s serialization.
163
+ const resultStr = typeof result === 'string' ? result : JSON.stringify(result);
164
+ this.loopReason = `Tool "${toolName}" returned errors ${this.consecutiveErrors} consecutive times. Last result: ${resultStr.slice(0, 200)}`;
162
165
  return true;
163
166
  }
164
167
  } else {
@@ -198,6 +201,40 @@ export class ToolCallLoopDetector {
198
201
  * const result = await runner.run('Check all team statuses');
199
202
  * ```
200
203
  */
204
+ /**
205
+ * Thrown when a single agent run exceeds its wall-clock budget.
206
+ *
207
+ * Distinct class (rather than a message-sniffed "timeout" string) because
208
+ * {@link AgentRunnerService.isRecoverableError} treats anything mentioning
209
+ * "timeout" as retryable. A hard-budget expiry is deliberately NOT retryable:
210
+ * the budget covers the whole run including retries, so re-entering the retry
211
+ * loop would multiply the very stall the budget exists to bound.
212
+ *
213
+ * @example
214
+ * ```typescript
215
+ * try { await runner.run(msg); }
216
+ * catch (e) { if (e instanceof AgentRunTimeoutError) recycleAgent(); }
217
+ * ```
218
+ */
219
+ export class AgentRunTimeoutError extends Error {
220
+ /** Wall-clock budget that was exceeded, in milliseconds. */
221
+ readonly timeoutMs: number;
222
+
223
+ /**
224
+ * @param timeoutMs - The budget that was exceeded, in milliseconds
225
+ * @param modelId - Model the run was using, for operator diagnosis
226
+ */
227
+ constructor(timeoutMs: number, modelId: string) {
228
+ super(
229
+ `Agent run exceeded its ${timeoutMs}ms wall-clock budget (model: ${modelId}) `
230
+ + `and was hard-aborted. The model connection stalled without delivering a `
231
+ + `response — see CREWLY_AGENT_MESSAGE_TIMEOUT_MS to tune the budget.`,
232
+ );
233
+ this.name = 'AgentRunTimeoutError';
234
+ this.timeoutMs = timeoutMs;
235
+ }
236
+ }
237
+
201
238
  /** Function type for generateText — used for dependency injection in tests */
202
239
  type GenerateTextFn = (opts: Record<string, unknown>) => Promise<Record<string, unknown>>;
203
240
 
@@ -824,19 +861,73 @@ export class AgentRunnerService {
824
861
  externalAbortSignal.addEventListener('abort', () => runAbort.abort(), { once: true });
825
862
  }
826
863
 
864
+ // Wall-clock budget for the ENTIRE run (all retries included).
865
+ //
866
+ // Without this, a model connection that opens but then goes silent — no
867
+ // bytes, no FIN, no error — hangs `await streamResult` forever: undici
868
+ // applies no idle deadline to a streaming body. That stalled promise wedges
869
+ // processQueue permanently (`this.processing` stays true), so every later
870
+ // message queues behind it and the agent goes silently catatonic: the
871
+ // process is alive and heartbeating, but no run ever completes and no error
872
+ // is ever raised. Bounding the run converts that invisible hang into a
873
+ // normal rejection the caller can log, surface, and recover from.
874
+ const timeoutMs = this.resolveRunTimeoutMs();
875
+ let hardTimer: ReturnType<typeof setTimeout> | undefined;
876
+ let softTimer: ReturnType<typeof setTimeout> | undefined;
877
+
827
878
  try {
828
879
  // If a test override is set, use generateText path (backward compatible)
829
- if (this._generateTextFn) {
830
- return await this.executeRunWithGenerateText(tools, runAbort.signal);
831
- }
880
+ const runPromise = this._generateTextFn
881
+ ? this.executeRunWithGenerateText(tools, runAbort.signal)
882
+ // Production path: streamText for real-time feedback
883
+ : this.executeRunWithStreamText(tools, runAbort.signal);
884
+
885
+ // The loser of the race stays pending. Attach a no-op catch so its
886
+ // eventual abort-rejection is never an unhandled rejection (which
887
+ // crashes the agent process under Node's default policy).
888
+ runPromise.catch(() => { /* settled by the race below */ });
889
+
890
+ const timeoutPromise = new Promise<never>((_, reject) => {
891
+ softTimer = setTimeout(() => {
892
+ console.warn(
893
+ `[AgentRunner] Run exceeded soft warning threshold `
894
+ + `(${CREWLY_AGENT_DEFAULTS.MESSAGE_SOFT_WARNING_MS}ms) — still waiting on the model.`,
895
+ );
896
+ }, CREWLY_AGENT_DEFAULTS.MESSAGE_SOFT_WARNING_MS);
897
+
898
+ hardTimer = setTimeout(() => {
899
+ // Abort first so the in-flight HTTP request and any tool work are
900
+ // torn down, then reject — a well-behaved provider unwinds on the
901
+ // signal, and a wedged one no longer holds the queue hostage.
902
+ runAbort.abort();
903
+ reject(new AgentRunTimeoutError(timeoutMs, this.config.model.modelId));
904
+ }, timeoutMs);
905
+ });
832
906
 
833
- // Production path: streamText for real-time feedback
834
- return await this.executeRunWithStreamText(tools, runAbort.signal);
907
+ return await Promise.race([runPromise, timeoutPromise]);
835
908
  } finally {
909
+ if (hardTimer) clearTimeout(hardTimer);
910
+ if (softTimer) clearTimeout(softTimer);
836
911
  this.currentRunAbort = null;
837
912
  }
838
913
  }
839
914
 
915
+ /**
916
+ * Resolve the wall-clock budget for one run.
917
+ *
918
+ * Per-model overrides win over the global default: reasoning models emit
919
+ * chain-of-thought tokens far slower than plain content, so a budget tuned
920
+ * for a chat model would kill them mid-thought.
921
+ *
922
+ * @returns Budget in milliseconds for the currently configured model
923
+ */
924
+ private resolveRunTimeoutMs(): number {
925
+ return (
926
+ CREWLY_AGENT_DEFAULTS.MODEL_TIMEOUT_MS[this.config.model.modelId]
927
+ ?? CREWLY_AGENT_DEFAULTS.MESSAGE_TIMEOUT_MS
928
+ );
929
+ }
930
+
840
931
  /**
841
932
  * Check if an error is recoverable and eligible for automatic retry.
842
933
  *
@@ -850,6 +941,9 @@ export class AgentRunnerService {
850
941
  */
851
942
  private isRecoverableError(error: unknown): boolean {
852
943
  if (!(error instanceof Error)) return false;
944
+ // Hard-budget expiry is terminal by construction: the budget spans the
945
+ // whole run, so retrying would just re-enter the stall it bounded.
946
+ if (error instanceof AgentRunTimeoutError) return false;
853
947
  const msg = error.message.toLowerCase();
854
948
  const statusMatch = msg.match(/\b(429|5\d{2})\b/);
855
949
  if (statusMatch) return true;
@@ -2186,6 +2186,25 @@ describe('Tool Registry', () => {
2186
2186
  expect(validateFn('rm -rf ./dist')).toBeNull();
2187
2187
  expect(validateFn('rm -rf node_modules')).toBeNull();
2188
2188
  });
2189
+
2190
+ it('should signal the block is permanent so the model stops retrying', () => {
2191
+ // A blocked command is a policy decision, not a transient failure. When
2192
+ // the message read like a retryable error the model retried variations
2193
+ // until the loop detector aborted the whole run, so the wording carries
2194
+ // real behavioural weight and is worth pinning.
2195
+ const reason = validateFn('kill -9 $$');
2196
+
2197
+ expect(reason).not.toBeNull();
2198
+ expect(reason).toMatch(/permanently blocked/i);
2199
+ expect(reason).toMatch(/NOT a transient error/i);
2200
+ expect(reason).toMatch(/do NOT retry/i);
2201
+ });
2202
+
2203
+ it('should still identify which pattern caused the block', () => {
2204
+ // The matched pattern stays in the message: without it the model cannot
2205
+ // tell which part of a compound command was rejected.
2206
+ expect(validateFn('shutdown -h now')).toMatch(/shutdown/);
2207
+ });
2189
2208
  });
2190
2209
 
2191
2210
  describe('bash_exec tool', () => {
@@ -79,7 +79,10 @@ export const APPROVAL_REQUIRED_BASH_PATTERNS: Array<{ pattern: RegExp; label: st
79
79
  export function validateBashCommand(command: string): string | null {
80
80
  for (const pattern of BLOCKED_COMMAND_PATTERNS) {
81
81
  if (pattern.test(command)) {
82
- return `Command blocked by security policy: matches forbidden pattern ${pattern}`;
82
+ // Phrased to signal the model this is a PERMANENT policy block, not a
83
+ // transient failure — otherwise it retries variations until the loop
84
+ // detector aborts the whole run.
85
+ return `Command permanently blocked by security policy (matched ${pattern}). This is NOT a transient error — do NOT retry this or a variation. Achieve the goal a different way (use a Crewly API/tool, or skip this step).`;
83
86
  }
84
87
  }
85
88
  return null;
@@ -540,6 +540,15 @@ export const CREWLY_AGENT_DEFAULTS = {
540
540
  /** Soft warning threshold in milliseconds — logs a warning but does not kill the request.
541
541
  * Override via CREWLY_AGENT_MESSAGE_SOFT_WARNING_MS env var. */
542
542
  MESSAGE_SOFT_WARNING_MS: Number(process.env.CREWLY_AGENT_MESSAGE_SOFT_WARNING_MS) || 240_000,
543
+ /** Grace period added on top of the agent's own run budget before the PARENT
544
+ * gives up waiting on the child's IPC reply.
545
+ *
546
+ * The child enforces the run budget itself and reports a clean error, so in
547
+ * normal operation the parent timer never fires. It exists for the case the
548
+ * child can no longer report at all — blocked event loop, broken stdout pipe,
549
+ * SIGSTOP. The grace keeps the parent from racing the child and mislabelling
550
+ * a normal in-budget timeout as a wedged runtime. */
551
+ IPC_RUN_TIMEOUT_GRACE_MS: 30_000,
543
552
  /** Cooldown period in milliseconds after rate limit retries are exhausted */
544
553
  RATE_LIMIT_COOLDOWN_MS: 300_000,
545
554
  /** Maximum number of automatic retries for recoverable errors (429, 5xx, network) */