@nowcrew/daemon 0.6.88-beta.17 → 0.6.88-beta.19

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -264,6 +264,7 @@ export const ExecutionRejectedSchema = z.object({
264
264
  reason: RejectionReasonSchema,
265
265
  message: z.string().optional(),
266
266
  at: TimestampSchema,
267
+ preAdmissionProof: z.object({ specHash: z.string().regex(/^[a-f0-9]{64}$/u) }).strict().optional(),
267
268
  }).strict();
268
269
  export const ExecutionStartedSchema = z.object({
269
270
  type: z.literal("execution:started"),
@@ -408,7 +408,8 @@ export async function runExecution(config, input, dependencies) {
408
408
  ? { permission: effectivePermission(spec, config) }
409
409
  : admission(spec, config, dependencies, initialAt);
410
410
  if ("rejected" in checked) {
411
- const frame = ExecutionRejectedSchema.parse(boundExecutionFrame(checked.rejected, config.executionLimits.maxEventBytes));
411
+ const proof = await dependencies.confirmPreAdmission?.(checked.rejected).catch(() => false);
412
+ const frame = ExecutionRejectedSchema.parse(boundExecutionFrame({ ...checked.rejected, ...(proof ? { preAdmissionProof: { specHash } } : {}) }, config.executionLimits.maxEventBytes));
412
413
  await dependencies.report(frame);
413
414
  return { kind: "rejected", frame };
414
415
  }
@@ -1,3 +1,4 @@
1
+ import { retryHarnessProbe } from "./runtimes/harness-probe-retry.js";
1
2
  /**
2
3
  * 采集本机信息上报给控制面 (machine:hello):hostname / os / daemon 版本 / 已装 runtimes。
3
4
  * runtimes 探测靠 `which <bin>`(win32 用 `where`),只报真实可执行的 CLI(用于展示 Detected Runtimes)。
@@ -16,7 +17,7 @@ import { createRequire } from "node:module";
16
17
  import { dirname, resolve } from "node:path";
17
18
  import { executableRuntimes } from "./runtime-capabilities.js";
18
19
  import { executionBackendCapability } from "./execution-backend.js";
19
- import { probeKimiAcp } from "./runtimes/kimi-acp-runner.js";
20
+ import { probeKimiAcp, AcpProbeCleanupError } from "./runtimes/kimi-acp-runner.js";
20
21
  import { probeHermesAcp } from "./runtimes/hermes.js";
21
22
  import { probeOpenCodeRun } from "./runtimes/opencode.js";
22
23
  import { RUNTIME_HEALTH_PROBE_CAPABILITY } from "./runtime-health.js";
@@ -132,7 +133,9 @@ async function supportsDeepSeekHarnessAcp(signal, probeId) {
132
133
  try {
133
134
  return await probeDeepSeekHarnessAcp({ bin: deepSeekHarnessLauncher().bin, ...(probeId ? { probeId } : {}), ...(signal ? { signal } : {}) });
134
135
  }
135
- catch {
136
+ catch (error) {
137
+ if (error instanceof AcpProbeCleanupError)
138
+ throw error;
136
139
  return false;
137
140
  }
138
141
  }
@@ -162,7 +165,9 @@ export async function detectExecutionRuntimes(installed = detectRuntimes(), kimi
162
165
  present.includes("kimi") ? probe("kimi", kimiAcpProbe) : false,
163
166
  present.includes("hermes") ? probe("hermes", hermesAcpProbe) : false,
164
167
  present.includes("opencode") ? probe("opencode", openCodeProbe) : false,
165
- present.includes("deepseek-harness") ? probe("deepseek-harness", deepSeekHarnessProbe) : false,
168
+ present.includes("deepseek-harness")
169
+ ? retryHarnessProbe(() => probe("deepseek-harness", deepSeekHarnessProbe), signal, probeId)
170
+ : false,
166
171
  ]);
167
172
  return executableRuntimes(present).filter((runtime) => (runtime === "kimi" ? kimiReady
168
173
  : runtime === "hermes" ? hermesReady
@@ -0,0 +1,64 @@
1
+ import { ExecutionCompletedSchema, ExecutionRejectedSchema } from "./execution-protocol.js";
2
+ import { hashExecutionSpec } from "./execution-runner.js";
3
+ import { dslog } from "./slog.js";
4
+ /** Queue rejection is admission work too: own the ID before any asynchronous journal write. */
5
+ export async function rejectQueuedAdmission(context, spec, executionClass, facts) {
6
+ const id = spec.executionId;
7
+ const hash = hashExecutionSpec(spec);
8
+ if (context.knownExecutionHashes.has(id) || context.executionRuns.has(id)) {
9
+ // A same-ID request won while the caller awaited journal.get. Its journal may
10
+ // still be pending; silence lets its real result arrive instead of rejecting it.
11
+ const existing = await context.journal.get(id);
12
+ if (existing?.specHash === hash)
13
+ context.replay(existing);
14
+ return;
15
+ }
16
+ context.knownExecutionHashes.set(id, hash);
17
+ const rejected = ExecutionRejectedSchema.parse({
18
+ type: "execution:rejected", protocolVersion: 1, executionId: id,
19
+ reason: "resource_limit", message: "Local machine execution queue is full", at: new Date().toISOString(),
20
+ });
21
+ dslog("execution.machine_queue_rejected", "机器执行队列已满", {
22
+ level: "WARN", execution_id: id, agent_handle: spec.agent.handle, execution_class: executionClass, ...facts,
23
+ });
24
+ const rejecting = persistPreAdmissionRejection(context.journal, spec, rejected)
25
+ .then(async (entry) => {
26
+ if (entry)
27
+ await context.report({ ...rejected, preAdmissionProof: { specHash: hash } });
28
+ else {
29
+ const existing = await context.journal.get(id);
30
+ if (existing?.specHash === hash)
31
+ context.replay(existing);
32
+ }
33
+ })
34
+ .catch(error => {
35
+ // A failed durable write has no safe rejection proof; a restart/sync will
36
+ // reconcile an accepted tombstone if this admission created one.
37
+ dslog("execution.admission_rejection_failed", "准入拒绝未取得持久停止证明", {
38
+ level: "ERROR", execution_id: id, error_message: String(error),
39
+ });
40
+ })
41
+ .finally(() => {
42
+ if (context.executionRuns.get(id) === rejecting) {
43
+ context.executionRuns.delete(id);
44
+ context.knownExecutionHashes.delete(id);
45
+ }
46
+ });
47
+ context.executionRuns.set(id, rejecting);
48
+ await rejecting;
49
+ }
50
+ /** The caller synchronously owns this admission ID. Never complete somebody else's record. */
51
+ export async function persistPreAdmissionRejection(journal, spec, rejection) {
52
+ const accepted = await journal.accept(spec.executionId, hashExecutionSpec(spec), {
53
+ runtime: spec.runtime.name, ...(spec.runtime.model ? { model: spec.runtime.model } : {}), resumed: false,
54
+ });
55
+ if (accepted.kind !== "created")
56
+ return null;
57
+ return journal.complete(spec.executionId, ExecutionCompletedSchema.parse({
58
+ type: "execution:completed", protocolVersion: 1, executionId: spec.executionId,
59
+ outcome: "failed", errorCode: rejection.reason, errorMessage: rejection.message ?? "Admission rejected before runtime start",
60
+ runtime: spec.runtime.name, ...(spec.runtime.model ? { model: spec.runtime.model } : {}), resumed: false,
61
+ startedAt: accepted.acceptedAt,
62
+ finishedAt: new Date(Math.max(Date.now(), Date.parse(accepted.acceptedAt))).toISOString(),
63
+ }));
64
+ }
@@ -4,7 +4,7 @@ import { mkdtemp, rm } from "node:fs/promises";
4
4
  import { tmpdir } from "node:os";
5
5
  import { join } from "node:path";
6
6
  import { prepareHarnessProfile } from "./deepseek-harness-profile.js";
7
- import { probeKimiAcp } from "./kimi-acp-runner.js";
7
+ import { probeKimiAcp, AcpProbeCleanupError } from "./kimi-acp-runner.js";
8
8
  export const DEEPSEEK_HARNESS_CONFIG_ENV = "NOWCREW_DEEPSEEK_HARNESS_CONFIG";
9
9
  export const DEEPSEEK_HARNESS_PATCH_ENV = "NOWCREW_DEEPSEEK_HARNESS_PATCH";
10
10
  export function deepSeekHarnessLauncher(env = process.env) {
@@ -61,7 +61,9 @@ export async function probeDeepSeekHarnessAcp(options, spawnProcess = spawn) {
61
61
  launchCwd: root,
62
62
  }, spawnProcess);
63
63
  }
64
- catch {
64
+ catch (error) {
65
+ if (error instanceof AcpProbeCleanupError)
66
+ throw error;
65
67
  return false;
66
68
  }
67
69
  finally {
@@ -0,0 +1,42 @@
1
+ import { dslog } from "../slog.js";
2
+ const RETRY_DELAYS_MS = [1_000, 2_000];
3
+ /** 只延迟当前启动探测,不设后台定时任务;stop 立即清除等待。 */
4
+ function waitForRetry(ms, signal) {
5
+ if (signal?.aborted)
6
+ return Promise.resolve(false);
7
+ return new Promise((resolve) => {
8
+ const finish = (ready) => {
9
+ clearTimeout(timer);
10
+ signal?.removeEventListener("abort", onAbort);
11
+ resolve(ready);
12
+ };
13
+ const onAbort = () => finish(false);
14
+ const timer = setTimeout(() => finish(true), ms);
15
+ signal?.addEventListener("abort", onAbort, { once: true });
16
+ });
17
+ }
18
+ /** 成功才公布能力;每次 probe 返回前已完成进程清理,异常保持原有传播。 */
19
+ export async function retryHarnessProbe(probe, signal, probeId) {
20
+ for (let attempt = 1; attempt <= RETRY_DELAYS_MS.length + 1; attempt++) {
21
+ if (signal?.aborted)
22
+ return false;
23
+ const ready = await probe();
24
+ dslog("runtime_probe.harness_attempt", "Harness 启动探测尝试结束", {
25
+ probe_id: probeId, runtime: "deepseek-harness", attempt, ready,
26
+ aborted: signal?.aborted ?? false,
27
+ });
28
+ if (signal?.aborted)
29
+ return false;
30
+ if (ready)
31
+ return true;
32
+ const delayMs = RETRY_DELAYS_MS[attempt - 1];
33
+ if (delayMs === undefined)
34
+ return false;
35
+ dslog("runtime_probe.harness_retry", "延迟重试 Harness 启动探测", {
36
+ probe_id: probeId, runtime: "deepseek-harness", next_attempt: attempt + 1, delay_ms: delayMs,
37
+ });
38
+ if (!await waitForRetry(delayMs, signal))
39
+ return false;
40
+ }
41
+ return false;
42
+ }
@@ -196,6 +196,9 @@ async function stopChild(child) {
196
196
  await closed;
197
197
  }
198
198
  }
199
+ export class AcpProbeCleanupError extends Error {
200
+ constructor() { super("ACP probe process cleanup unconfirmed"); this.name = "AcpProbeCleanupError"; }
201
+ }
199
202
  /** Probe the ACP transport without starting a session or forcing an optional interactive login flow. */
200
203
  export async function probeKimiAcp(options, spawnProcess = spawn) {
201
204
  const provider = options.provider ?? "kimi";
@@ -280,6 +283,9 @@ export async function probeKimiAcp(options, spawnProcess = spawn) {
280
283
  await stopChild(child);
281
284
  cleanupFinished = true;
282
285
  }
286
+ catch {
287
+ throw new AcpProbeCleanupError();
288
+ }
283
289
  finally {
284
290
  dslog("execution.acp_probe_finished", "ACP 探测结束", { ...fields(),
285
291
  cleanup_finished: cleanupFinished, exit_code: child.exitCode, exit_signal: child.signalCode,
package/dist/serve.js CHANGED
@@ -3,6 +3,7 @@
3
3
  * 断线指数退避重连。一个 agent:start 进来就跑一次现有的 runAgent。
4
4
  */
5
5
  import { WebSocket } from "ws";
6
+ import { persistPreAdmissionRejection, rejectQueuedAdmission } from "./pre-admission-rejection.js";
6
7
  import { join } from "node:path";
7
8
  import { randomUUID } from "node:crypto";
8
9
  import { dslog, setSlogDefaults, drainSpool, flushSlog } from "./slog.js";
@@ -210,6 +211,11 @@ export function serve(config, opts = {}) {
210
211
  return;
211
212
  }
212
213
  safeExecutionSend(frame);
214
+ if (frame.type === "execution:rejected" && frame.preAdmissionProof) {
215
+ const entry = await executionJournal.get(frame.executionId);
216
+ if (entry?.completion)
217
+ completionRetransmitter.track(entry.completion);
218
+ }
213
219
  };
214
220
  const replayExecutionTelemetry = async () => {
215
221
  const frames = await executionTelemetry.replay();
@@ -530,18 +536,8 @@ export function serve(config, opts = {}) {
530
536
  : spec.context.scheduledRunId === undefined ? "normal" : "scheduled";
531
537
  let reservation = sharedSlots.reserve(spec.agent.handle, "execution", executionClass);
532
538
  if (!reservation.accepted) {
533
- dslog("execution.machine_queue_rejected", "机器执行队列已满", {
534
- level: "WARN",
535
- execution_id: spec.executionId,
536
- agent_handle: spec.agent.handle,
537
- execution_class: executionClass,
538
- ...reservation.facts,
539
- });
540
- safeExecutionSend(ExecutionRejectedSchema.parse({
541
- type: "execution:rejected", protocolVersion: 1, executionId: spec.executionId,
542
- reason: "resource_limit", message: "Local machine execution queue is full",
543
- at: new Date().toISOString(),
544
- }));
539
+ await rejectQueuedAdmission({ journal: executionJournal, knownExecutionHashes, executionRuns,
540
+ report: reportExecutionFrame, replay: sendJournalStatus }, spec, executionClass, reservation.facts);
545
541
  return;
546
542
  }
547
543
  knownExecutionHashes.set(spec.executionId, hash);
@@ -711,6 +707,13 @@ export function serve(config, opts = {}) {
711
707
  ...reservation.facts,
712
708
  },
713
709
  report: reportExecutionFrame,
710
+ confirmPreAdmission: async (rejection) => {
711
+ const prior = await executionJournal.get(spec.executionId);
712
+ if (prior !== null || stopped || knownExecutionHashes.get(spec.executionId) !== hash
713
+ || executionRuns.get(spec.executionId) !== execution)
714
+ return false;
715
+ return await persistPreAdmissionRejection(executionJournal, spec, rejection) !== null;
716
+ },
714
717
  startupGate: runtimeStartupGate,
715
718
  startupTimeoutMs: config.executionLimits.startupTimeoutMs,
716
719
  onRuntimePhase: (phase, runtime) => {
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@nowcrew/daemon",
3
- "version": "0.6.88-beta.17",
3
+ "version": "0.6.88-beta.19",
4
4
  "type": "module",
5
5
  "description": "crew daemon — 运行在用户机器:拉起/管理 agent 进程,注入 crew CLI,归一化 runtime 事件",
6
6
  "license": "Apache-2.0",