@nowcrew/daemon 0.6.63 → 0.6.64
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -8
- package/dist/execution-runner.js +13 -34
- package/package.json +1 -1
package/README.md
CHANGED
|
@@ -380,6 +380,7 @@ CREW_EXECUTION_MAX_STARTING_TOTAL=10
|
|
|
380
380
|
CREW_EXECUTION_MAX_STARTING_PER_RUNTIME=10
|
|
381
381
|
CREW_EXECUTION_START_GAP_MS=500
|
|
382
382
|
CREW_EXECUTION_STARTUP_TIMEOUT_MS=120000
|
|
383
|
+
# Retained for compatibility; accepted executions wait indefinitely for a slot.
|
|
383
384
|
CREW_EXECUTION_MAX_QUEUE_WAIT_MS=300000
|
|
384
385
|
CREW_EXECUTION_MAX_FIRST_OUTPUT_WAIT_MS=120000
|
|
385
386
|
CREW_EXECUTION_FIRST_OUTPUT_GRACE_MS=180000
|
|
@@ -391,14 +392,14 @@ protocol-v1 and legacy work.
|
|
|
391
392
|
Runtime startup is a separate FIFO gate: by default up to ten Claude, Codex, or Kimi processes may be
|
|
392
393
|
initializing at a time, and launches are spaced by 500ms. A process leaves the startup gate
|
|
393
394
|
only after its runtime-specific ready event; startup timeout cancels the owned process tree.
|
|
394
|
-
An accepted execution
|
|
395
|
-
the daemon waits up to two minutes for the first model event, with a
|
|
396
|
-
slow providers. When
|
|
397
|
-
terminal result is written to the execution journal before it is
|
|
398
|
-
replayable instead of being silently dropped.
|
|
399
|
-
most twice (three total attempts). Each retry releases its current
|
|
400
|
-
new one, which puts it behind work already waiting in the queue. A
|
|
401
|
-
not requeued; other failures are not automatically retried.
|
|
395
|
+
An accepted execution waits indefinitely for a local and host slot unless it is explicitly cancelled.
|
|
396
|
+
After runtime readiness, the daemon waits up to two minutes for the first model event, with a
|
|
397
|
+
three-minute grace window for slow providers. When the first-output liveness limit expires, the
|
|
398
|
+
supervisor is stopped and the terminal result is written to the execution journal before it is
|
|
399
|
+
reported, so the execution remains replayable instead of being silently dropped. First-output
|
|
400
|
+
timeouts are retried locally at most twice (three total attempts). Each retry releases its current
|
|
401
|
+
machine/host slot and reserves a new one, which puts it behind work already waiting in the queue. A
|
|
402
|
+
third timeout is terminal and is not requeued; other failures are not automatically retried.
|
|
402
403
|
|
|
403
404
|
Local policy can reduce server-requested access and limits; it cannot grant more access than requested.
|
|
404
405
|
Provider credentials and configured environment are prepared locally and never carried in execution
|
package/dist/execution-runner.js
CHANGED
|
@@ -299,12 +299,6 @@ class ExecutionCancelledError extends Error {
|
|
|
299
299
|
this.name = "ExecutionCancelledError";
|
|
300
300
|
}
|
|
301
301
|
}
|
|
302
|
-
class ExecutionQueueTimeoutError extends Error {
|
|
303
|
-
constructor(timeoutMs) {
|
|
304
|
-
super(`Execution queue wait exceeded ${timeoutMs}ms`);
|
|
305
|
-
this.name = "ExecutionQueueTimeoutError";
|
|
306
|
-
}
|
|
307
|
-
}
|
|
308
302
|
async function cancellable(promise, cancellation) {
|
|
309
303
|
if (cancellation === undefined)
|
|
310
304
|
return promise;
|
|
@@ -315,24 +309,9 @@ async function cancellable(promise, cancellation) {
|
|
|
315
309
|
cancellation.requested.then(() => { throw new ExecutionCancelledError(); }),
|
|
316
310
|
]);
|
|
317
311
|
}
|
|
318
|
-
async function cancellableWithTimeout(promise, timeoutMs, cancellation, timeoutError) {
|
|
319
|
-
let timer;
|
|
320
|
-
try {
|
|
321
|
-
return await Promise.race([
|
|
322
|
-
cancellable(promise, cancellation),
|
|
323
|
-
new Promise((_resolve, reject) => {
|
|
324
|
-
timer = setTimeout(() => reject(timeoutError), timeoutMs);
|
|
325
|
-
}),
|
|
326
|
-
]);
|
|
327
|
-
}
|
|
328
|
-
finally {
|
|
329
|
-
if (timer !== undefined)
|
|
330
|
-
clearTimeout(timer);
|
|
331
|
-
}
|
|
332
|
-
}
|
|
333
312
|
function failedCompletion(spec, error, startedAt, finishedAt) {
|
|
334
313
|
const message = error instanceof Error ? error.message : String(error);
|
|
335
|
-
const errorCode = error instanceof
|
|
314
|
+
const errorCode = error instanceof HostReservationCancelledError
|
|
336
315
|
? "queue_timeout"
|
|
337
316
|
: error instanceof ProjectContextUnavailableError
|
|
338
317
|
? error.code
|
|
@@ -519,7 +498,9 @@ export async function runExecution(config, input, dependencies) {
|
|
|
519
498
|
});
|
|
520
499
|
try {
|
|
521
500
|
if (dependencies.slot !== undefined) {
|
|
522
|
-
|
|
501
|
+
// Queueing is not a liveness failure. Keep the execution durable until its
|
|
502
|
+
// machine/host slot is granted or the server explicitly cancels it.
|
|
503
|
+
await cancellable(dependencies.slot.ready, dependencies.cancellation);
|
|
523
504
|
}
|
|
524
505
|
let projectContext;
|
|
525
506
|
let sessionContextFingerprint;
|
|
@@ -562,17 +543,15 @@ export async function runExecution(config, input, dependencies) {
|
|
|
562
543
|
if (cancel !== null && timeout === undefined && !dependencies.cancellation?.isRequested()) {
|
|
563
544
|
const firstOutputDeadline = config.executionLimits.maxFirstOutputWaitMs
|
|
564
545
|
+ config.executionLimits.firstOutputGraceMs;
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
}, firstOutputDeadline);
|
|
575
|
-
}
|
|
546
|
+
firstOutputTimer = setTimeout(() => {
|
|
547
|
+
firstOutputTimedOut = true;
|
|
548
|
+
try {
|
|
549
|
+
void cancel().catch(rejectCancellationFailure);
|
|
550
|
+
}
|
|
551
|
+
catch (error) {
|
|
552
|
+
rejectCancellationFailure(error);
|
|
553
|
+
}
|
|
554
|
+
}, firstOutputDeadline);
|
|
576
555
|
timeout = setTimeout(() => {
|
|
577
556
|
timedOut = true;
|
|
578
557
|
try {
|