claude-code-session-manager 0.75.3 → 0.77.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (164) hide show
  1. package/dist/assets/{AgentLibrary-CzQqcObq.js → AgentLibrary-B2ie8bbw.js} +2 -2
  2. package/dist/assets/{DataModel-Bj_WlLz8.js → DataModel-BIJPYw32.js} +1 -1
  3. package/dist/assets/{History-DnSi_OHm.js → History-CeY6dk9S.js} +2 -2
  4. package/dist/assets/{Hooks-0BB0dp3S.js → Hooks-BFH2ocKg.js} +2 -2
  5. package/dist/assets/{HostBilko-DHpwwsLQ.js → HostBilko-36gj9wLz.js} +1 -1
  6. package/dist/assets/{Library-CaJVqVvi.js → Library-C-hBct39.js} +1 -1
  7. package/dist/assets/{ListDetail-C1W2HmC2.js → ListDetail-CNq64VWV.js} +1 -1
  8. package/dist/assets/{MarkdownEditor-5Ob9FW3z.js → MarkdownEditor-Bh3qt5-1.js} +1 -1
  9. package/dist/assets/{McpServers-JxCSfm1S.js → McpServers-DpGN0oyz.js} +1 -1
  10. package/dist/assets/{Memory-BDeqlqwH.js → Memory-D59hUjC4.js} +6 -6
  11. package/dist/assets/{Panel-Dh9ZHuEj.js → Panel-DCgbaoci.js} +1 -1
  12. package/dist/assets/{Permissions-DXy-CbEY.js → Permissions-DAmQ0DYV.js} +2 -2
  13. package/dist/assets/{Plugins-_n1Iuc8T.js → Plugins-Dyfgn6Is.js} +2 -2
  14. package/dist/assets/{ProvenanceBadge-BP_evfxE.js → ProvenanceBadge-BiYhPO1U.js} +1 -1
  15. package/dist/assets/SaveBar-RV7B6sOh.js +1 -0
  16. package/dist/assets/Scheduler-BPaNqx1b.js +14 -0
  17. package/dist/assets/{ScopeSwitcher-CAWzM6RI.js → ScopeSwitcher-P4mdLGNU.js} +1 -1
  18. package/dist/assets/{Settings-DRRozLyT.js → Settings-BL4vf5aX.js} +1 -1
  19. package/dist/assets/{SkillReferenceGraph-DGHDWlz4.js → SkillReferenceGraph-BRBDyi1_.js} +1 -1
  20. package/dist/assets/{Skills-D8L66eiX.js → Skills-BV08gDUH.js} +2 -2
  21. package/dist/assets/{SystemPrompt-CYtUsonD.js → SystemPrompt-CLftSsDw.js} +1 -1
  22. package/dist/assets/TagLibrary-Bp8jGsd5.js +1 -0
  23. package/dist/assets/{TiptapBody-B2hRgbPE.js → TiptapBody-jCpuB6E5.js} +1 -1
  24. package/dist/assets/{Toggle-BTwsbxam.js → Toggle-D2paA1xf.js} +1 -1
  25. package/dist/assets/{index-DijufvkJ.js → index-BDRSqBl3.js} +704 -704
  26. package/dist/assets/{index-CMLnzdZC.css → index-CYhdtisq.css} +1 -1
  27. package/dist/assets/{settingsSchema-D6wzxAi6.js → settingsSchema-6IOLjZZN.js} +1 -1
  28. package/dist/index.html +2 -2
  29. package/package.json +8 -2
  30. package/plugins/session-manager-dev/skills/develop/standards.md +1 -1
  31. package/scripts/lib/activeSessions.cjs +116 -6
  32. package/scripts/project-pages-logic/dist/logic.cjs +4709 -0
  33. package/scripts/render-project-pages/dist/renderer.cjs +18900 -0
  34. package/scripts/render-project-pages.cjs +70 -0
  35. package/scripts/scheduler-mcp-server.cjs +269 -96
  36. package/scripts/validate-project-pages-summary.cjs +62 -0
  37. package/src/main/__tests__/agentModelResolve.test.cjs +66 -0
  38. package/src/main/__tests__/epicStatusMirror.test.cjs +110 -0
  39. package/src/main/__tests__/health-delegation-chain.test.cjs +106 -0
  40. package/src/main/__tests__/prdAdminRoutes.test.cjs +295 -0
  41. package/src/main/__tests__/prdAgentType.test.cjs +103 -0
  42. package/src/main/__tests__/prdCreate.test.cjs +247 -0
  43. package/src/main/__tests__/prdFrontmatterAgentType.test.cjs +117 -0
  44. package/src/main/__tests__/prdFrontmatterQuietMachine.test.cjs +108 -0
  45. package/src/main/__tests__/projectHomeAdminRoutes.test.cjs +485 -0
  46. package/src/main/__tests__/projectPages.test.cjs +73 -1
  47. package/src/main/__tests__/rcaReport.test.cjs +54 -0
  48. package/src/main/__tests__/runVerify.test.cjs +94 -0
  49. package/src/main/__tests__/scheduler-autofix-select.test.cjs +58 -3
  50. package/src/main/__tests__/scheduler-bash-timeout-env.test.cjs +103 -0
  51. package/src/main/__tests__/scheduler-commit-guard-noop.test.cjs +41 -0
  52. package/src/main/__tests__/scheduler-effective-concurrency.test.cjs +10 -0
  53. package/src/main/__tests__/scheduler-foreign-wip-manifest.test.cjs +78 -0
  54. package/src/main/__tests__/scheduler-inplace-salvage.test.cjs +242 -0
  55. package/src/main/__tests__/scheduler-investigation-prompt.test.cjs +31 -0
  56. package/src/main/__tests__/scheduler-launch-failure.test.cjs +201 -0
  57. package/src/main/__tests__/scheduler-leftover-fields.test.cjs +52 -0
  58. package/src/main/__tests__/scheduler-looks-done.test.cjs +241 -0
  59. package/src/main/__tests__/scheduler-prd-persona-spawn.test.cjs +135 -0
  60. package/src/main/__tests__/scheduler-quiet-machine-lease.test.cjs +222 -0
  61. package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +207 -1
  62. package/src/main/__tests__/scheduler-shared-tree-guard.test.cjs +212 -0
  63. package/src/main/__tests__/scheduler-stranded-investigation.test.cjs +185 -0
  64. package/src/main/__tests__/scheduler-worktree-cap-defer.test.cjs +194 -0
  65. package/src/main/__tests__/seedAgentPersonas.test.cjs +75 -14
  66. package/src/main/__tests__/seedSchedulerMcp.test.cjs +66 -0
  67. package/src/main/__tests__/uniquePrdNumbers.test.cjs +14 -5
  68. package/src/main/bilkoHost.cjs +4 -3
  69. package/src/main/chatRunner.cjs +6 -1
  70. package/src/main/config.cjs +25 -33
  71. package/src/main/health.cjs +153 -2
  72. package/src/main/index.cjs +64 -5
  73. package/src/main/ipcSchemas.cjs +69 -1
  74. package/src/main/lib/__tests__/activeIndexRebuild.test.cjs +179 -0
  75. package/src/main/lib/__tests__/childWithLog.test.cjs +141 -0
  76. package/src/main/lib/__tests__/delegationReadiness.test.cjs +391 -42
  77. package/src/main/lib/__tests__/ephemeralCwd.test.cjs +91 -0
  78. package/src/main/lib/__tests__/epicWorktreeMint.test.cjs +5 -3
  79. package/src/main/lib/__tests__/fixChainDepth.test.cjs +40 -0
  80. package/src/main/lib/__tests__/gitWorktree.test.cjs +290 -5
  81. package/src/main/lib/__tests__/gitWorktreeSalvage.test.cjs +107 -0
  82. package/src/main/lib/__tests__/gitWorktreeSalvageDelta.test.cjs +153 -0
  83. package/src/main/lib/__tests__/jobWorktree.test.cjs +6 -4
  84. package/src/main/lib/__tests__/landedSinceRun.test.cjs +73 -0
  85. package/src/main/lib/__tests__/launchFailure.test.cjs +220 -0
  86. package/src/main/lib/__tests__/loadGate.test.cjs +159 -0
  87. package/src/main/lib/__tests__/mcpToolCatalog.test.cjs +102 -0
  88. package/src/main/lib/__tests__/opsOwnership.test.cjs +7 -0
  89. package/src/main/lib/__tests__/opsRootAbsoluteCwd.test.cjs +151 -0
  90. package/src/main/lib/__tests__/opsRootResolve.test.cjs +149 -0
  91. package/src/main/lib/__tests__/prdDeclaredPaths.test.cjs +82 -0
  92. package/src/main/lib/__tests__/projectRootResolve.test.cjs +148 -0
  93. package/src/main/lib/__tests__/queueHealth.test.cjs +58 -0
  94. package/src/main/lib/__tests__/quietMachineLease.test.cjs +39 -0
  95. package/src/main/lib/__tests__/reaperHelpers.test.cjs +133 -0
  96. package/src/main/lib/__tests__/schedulerBatchDepends.test.cjs +19 -9
  97. package/src/main/lib/__tests__/schedulerBatchFairness.test.cjs +213 -0
  98. package/src/main/lib/__tests__/schedulerBatchLaunchHold.test.cjs +125 -0
  99. package/src/main/lib/__tests__/schedulerBatchProjectCap.test.cjs +127 -0
  100. package/src/main/lib/__tests__/schedulerBatchQuietMachine.test.cjs +109 -0
  101. package/src/main/lib/__tests__/schedulerMcpServerHeadlessRefusal.test.cjs +71 -0
  102. package/src/main/lib/__tests__/schedulerMcpServerHelp.test.cjs +217 -0
  103. package/src/main/lib/__tests__/schedulerMcpServerProjectHome.test.cjs +350 -0
  104. package/src/main/lib/activeIndexMerge.cjs +15 -0
  105. package/src/main/lib/activeIndexRebuild.cjs +133 -0
  106. package/src/main/lib/agentModelResolve.cjs +58 -0
  107. package/src/main/lib/buildTarget.cjs +3 -2
  108. package/src/main/lib/childWithLog.cjs +69 -2
  109. package/src/main/lib/claudeBin.cjs +54 -1
  110. package/src/main/lib/crossProjectFeedback.cjs +8 -1
  111. package/src/main/lib/definitionOfDone.cjs +3 -2
  112. package/src/main/lib/delegationReadiness.cjs +514 -26
  113. package/src/main/lib/ephemeralCwd.cjs +78 -0
  114. package/src/main/lib/epicDelegationStats.cjs +2 -1
  115. package/src/main/lib/epicMint.cjs +17 -1
  116. package/src/main/lib/epicStatusMirror.cjs +95 -0
  117. package/src/main/lib/epicValidationHook.cjs +2 -1
  118. package/src/main/lib/epicWorktreeMint.cjs +5 -2
  119. package/src/main/lib/fixChainDepth.cjs +45 -0
  120. package/src/main/lib/gitWorktree.cjs +520 -21
  121. package/src/main/lib/jobWorktree.cjs +2 -0
  122. package/src/main/lib/landedSinceRun.cjs +55 -0
  123. package/src/main/lib/launchFailure.cjs +357 -0
  124. package/src/main/lib/loadGate.cjs +134 -0
  125. package/src/main/lib/mcpToolCatalog.cjs +370 -0
  126. package/src/main/lib/opsErrorLog.cjs +12 -1
  127. package/src/main/lib/opsOwnership.cjs +106 -0
  128. package/src/main/lib/prdAdminRoutes.cjs +43 -3
  129. package/src/main/lib/prdAgentType.cjs +84 -0
  130. package/src/main/lib/prdCreate.cjs +103 -15
  131. package/src/main/lib/prdDeclaredPaths.cjs +70 -0
  132. package/src/main/lib/prdFrontmatter.cjs +17 -3
  133. package/src/main/lib/prdLocations.cjs +13 -6
  134. package/src/main/lib/projectHomeAdminRoutes.cjs +402 -0
  135. package/src/main/lib/projectPageSummarySchema.cjs +181 -0
  136. package/src/main/lib/projectRootResolve.cjs +134 -0
  137. package/src/main/lib/promptSessionSchema.cjs +7 -0
  138. package/src/main/lib/queueHealth.cjs +38 -0
  139. package/src/main/lib/queueStore.cjs +40 -7
  140. package/src/main/lib/quietMachineLease.cjs +48 -0
  141. package/src/main/lib/rcaReport.cjs +54 -4
  142. package/src/main/lib/reaperHelpers.cjs +64 -1
  143. package/src/main/lib/scheduleJobSchema.cjs +31 -0
  144. package/src/main/lib/scheduleJobTransitions.cjs +6 -2
  145. package/src/main/lib/schedulerBatch.cjs +301 -55
  146. package/src/main/lib/schedulerConfig.cjs +99 -0
  147. package/src/main/projectBrief.cjs +3 -2
  148. package/src/main/projectPages.cjs +162 -3
  149. package/src/main/promptSessionTranscript.cjs +0 -0
  150. package/src/main/pty.cjs +5 -0
  151. package/src/main/queueOps.cjs +15 -8
  152. package/src/main/runVerify.cjs +50 -9
  153. package/src/main/scheduler/prdParser.cjs +18 -1
  154. package/src/main/scheduler.cjs +1701 -130
  155. package/src/main/seedAgentPersonas.cjs +62 -21
  156. package/src/main/seedSchedulerMcp.cjs +58 -4
  157. package/src/main/templates/project-pages-catalog.json +741 -0
  158. package/src/main/templates/project-pages-pipeline.md +417 -0
  159. package/src/preload/api.d.ts +187 -3
  160. package/src/preload/index.cjs +9 -0
  161. package/src/seed/agents/project-home-builder.md +59 -0
  162. package/dist/assets/SaveBar-D-gCUx4n.js +0 -1
  163. package/dist/assets/Scheduler-Bpd4OGju.js +0 -14
  164. package/dist/assets/TagLibrary-E5CLeuVk.js +0 -1
@@ -62,6 +62,8 @@ module.exports = {
62
62
  createJobWorktree: gitWorktree.createJobWorktree,
63
63
  integrateJobBranch: gitWorktree.integrateJobBranch,
64
64
  cleanupJobWorktree: gitWorktree.cleanupJobWorktree,
65
+ salvageJobWorktreeDiff: gitWorktree.salvageJobWorktreeDiff,
66
+ salvageJobDirtyDelta: gitWorktree.salvageJobDirtyDelta,
65
67
  parseWorktreeListPorcelain: gitWorktree.parseWorktreeListPorcelain,
66
68
  reconcileWorktreesOnBoot: (cwds) => gitWorktree.reconcileWorktreesOnBoot(cwds, { kind: KIND }),
67
69
  // Test-only escape hatch for the in-memory concurrency counter.
@@ -0,0 +1,55 @@
1
+ 'use strict';
2
+
3
+ /**
4
+ * landedSinceRun.cjs — widened, path-scoped commit evidence for the reverify
5
+ * self-heal pass (PRD 1102).
6
+ *
7
+ * committedInWindow (scheduler.cjs) only sees commits inside
8
+ * [startedAt, finishedAt+60s] — a commit that lands later (a retry, a
9
+ * sibling run, a human) is invisible to it. landedSinceRun has no upper
10
+ * bound, but narrows the OTHER way that committedInWindow is dangerously
11
+ * broad: it is scoped to paths the PRD itself declares, so an unrelated
12
+ * commit elsewhere in the repo is not credited to this job (see
13
+ * scheduler.cjs's healRefusalReason for why repo-wide, unscoped evidence is
14
+ * not attribution).
15
+ *
16
+ * Pure git wrapper — no fetch, no scheduler state. Callers that need remote
17
+ * commits visible (e.g. a job that committed in a since-removed worktree)
18
+ * must call scheduler.cjs's fetchAllRefs(cwd) first, same as
19
+ * committedInWindow's own callers do.
20
+ */
21
+
22
+ const { execFile } = require('node:child_process');
23
+
24
+ const LANDED_SINCE_RUN_TIMEOUT_MS = 10_000;
25
+
26
+ /**
27
+ * Commits in `cwd` since `sinceIso` (no upper bound) that touch any of
28
+ * `paths`. Never throws — git-unavailable, a non-repo cwd, or an empty
29
+ * `paths` list all resolve to `[]` rather than fabricating evidence.
30
+ *
31
+ * @param {string} cwd
32
+ * @param {string} sinceIso
33
+ * @param {string[]} paths
34
+ * @param {{ timeoutMs?: number }} [opts]
35
+ * @returns {Promise<string[]>} full commit SHAs, newest first
36
+ */
37
+ function landedSinceRun(cwd, sinceIso, paths, { timeoutMs = LANDED_SINCE_RUN_TIMEOUT_MS } = {}) {
38
+ return new Promise((resolve) => {
39
+ if (!cwd || !sinceIso || !Array.isArray(paths) || paths.length === 0) {
40
+ resolve([]);
41
+ return;
42
+ }
43
+ execFile(
44
+ 'git',
45
+ ['-C', cwd, 'log', '--all', `--since=${sinceIso}`, '--format=%H', '--', ...paths],
46
+ { timeout: timeoutMs, windowsHide: true },
47
+ (err, stdout) => {
48
+ if (err) { resolve([]); return; }
49
+ resolve(String(stdout || '').trim().split('\n').filter(Boolean));
50
+ },
51
+ );
52
+ });
53
+ }
54
+
55
+ module.exports = { landedSinceRun, LANDED_SINCE_RUN_TIMEOUT_MS };
@@ -0,0 +1,357 @@
1
+ 'use strict';
2
+
3
+ /**
4
+ * launchFailure.cjs — detects a headless `claude -p` run that NEVER RAN, and
5
+ * models the per-persona launch circuit breaker the scheduler routes on.
6
+ *
7
+ * Incident (GitHub issue #11, 2026-09-02, macOS): the installed Claude CLI
8
+ * sent `thinking.type.enabled` to a model that only accepts
9
+ * `thinking.type.adaptive`. The API answered HTTP 400 on the very first
10
+ * request, so every job did zero turns, spent zero output tokens, wrote zero
11
+ * files, exited 1 in ~25 s — and was recorded as `failed` with `error: null`,
12
+ * indistinguishable from a real implementation failure. The auto-fix
13
+ * investigation probe then launched with the same CLI and died the same way.
14
+ * 12 of 41 transcripts in one project over a month were this signature.
15
+ *
16
+ * Two facts this module makes first-class:
17
+ *
18
+ * 1. A NON-RUN is not a failure of the PRD. `classifyLaunchFailure` is
19
+ * deliberately narrow — it only fires when the transcript's `result`
20
+ * event shows no real turn (num_turns ≤ 1 AND output_tokens = 0) AND the
21
+ * result text carries the CLI's literal `API Error:` prefix. Anything
22
+ * that did a turn, or failed without the API marker, is somebody else's
23
+ * classification (rate-limit, network, transient, verifier).
24
+ *
25
+ * 2. The environment is broken, not the job — so the scheduler must stop
26
+ * re-dispatching identical launches (each one is a wasted 25 s + a
27
+ * misleading `failed` row + a doomed investigation) while still
28
+ * self-healing the moment the environment is fixed. That is a circuit
29
+ * breaker keyed by the launch persona (`agentType` → model): closed
30
+ * (normal) → open (blocked, exponential backoff) → half-open (exactly one
31
+ * probe job goes through) → closed again on a real turn. A CLI version
32
+ * change (the actual fix for the incident: `claude update`) short-circuits
33
+ * the backoff so the queue resumes on the next tick, not the next hour.
34
+ *
35
+ * Pure and Electron-free: every function here takes plain values (a parsed
36
+ * result event, a block record, `now`) so the whole state machine is
37
+ * unit-testable without a spawn. scheduler.cjs owns the I/O around it.
38
+ */
39
+
40
+ const fs = require('node:fs');
41
+ const path = require('node:path');
42
+ const { readTail } = require('./fileTail.cjs');
43
+
44
+ const LAUNCH_FAILURE_KINDS = Object.freeze({
45
+ /** HTTP 400 naming a thinking/effort/config parameter the model rejects — the issue-#11 signature. */
46
+ MODEL_CONFIG_REJECTED: 'model_config_rejected',
47
+ /** Any other HTTP 400 on the first request (malformed request body, unsupported flag combo). */
48
+ BAD_REQUEST: 'bad_request',
49
+ /** HTTP 401/403 — the CLI's credentials are missing, expired, or lack access to the model. */
50
+ AUTH_FAILED: 'auth_failed',
51
+ /** HTTP 404 that names the model — the pinned `--model` does not exist for this account/CLI. */
52
+ MODEL_NOT_FOUND: 'model_not_found',
53
+ /** HTTP 5xx / 529 / "Overloaded" — the API itself is unavailable right now. */
54
+ API_OVERLOADED: 'api_overloaded',
55
+ /** Any other first-request API error. */
56
+ API_ERROR: 'api_error',
57
+ });
58
+
59
+ /** Tail bytes scanned for the `result` event — same budget classifyRunOutcome uses. */
60
+ const RESULT_TAIL_BYTES = 65536;
61
+
62
+ /**
63
+ * Parse the LAST `{"type":"result",...}` stream-json event out of a log tail.
64
+ * Returns a flat, typed summary or null when no result event is present
65
+ * (the process died before the harness could emit one — that is
66
+ * reaperHelpers' `no_result`, not a launch failure).
67
+ */
68
+ function parseResultEvent(text) {
69
+ if (!text) return null;
70
+ let last = null;
71
+ for (const line of String(text).split('\n')) {
72
+ const t = line.trim();
73
+ if (!t.startsWith('{') || !t.includes('"type":"result"')) continue;
74
+ try {
75
+ const obj = JSON.parse(t);
76
+ if (obj && obj.type === 'result') last = obj;
77
+ } catch { /* partial line at the tail boundary */ }
78
+ }
79
+ if (!last) return null;
80
+ const usage = last.usage && typeof last.usage === 'object' ? last.usage : {};
81
+ const num = (v) => (typeof v === 'number' && Number.isFinite(v) ? v : null);
82
+ return {
83
+ subtype: typeof last.subtype === 'string' ? last.subtype : '',
84
+ isError: last.is_error === true,
85
+ numTurns: num(last.num_turns),
86
+ outputTokens: num(usage.output_tokens),
87
+ inputTokens: num(usage.input_tokens),
88
+ apiErrorStatus: num(last.api_error_status),
89
+ totalCostUsd: num(last.total_cost_usd),
90
+ durationMs: num(last.duration_ms),
91
+ terminalReason: typeof last.terminal_reason === 'string' ? last.terminal_reason : null,
92
+ resultText: typeof last.result === 'string' ? last.result : '',
93
+ };
94
+ }
95
+
96
+ function readResultEvent(logPath) {
97
+ try {
98
+ return parseResultEvent(readTail(logPath, RESULT_TAIL_BYTES));
99
+ } catch {
100
+ return null;
101
+ }
102
+ }
103
+
104
+ /**
105
+ * Pull the human-readable message out of the CLI's `API Error: <status> <json>`
106
+ * text. The body nests unpredictably (`{"error":{"message":...}}`, or
107
+ * `{"detail":{"error":"<json string with message>"}}` as in issue #11), so
108
+ * this walks any `message`/`error` chain it can parse and falls back to the
109
+ * raw text, bounded.
110
+ */
111
+ function extractApiMessage(text) {
112
+ const raw = String(text || '').trim();
113
+ const jsonStart = raw.indexOf('{');
114
+ if (jsonStart >= 0) {
115
+ let node;
116
+ try { node = JSON.parse(raw.slice(jsonStart)); } catch { node = null; }
117
+ let depth = 0;
118
+ while (node && depth < 6) {
119
+ depth += 1;
120
+ if (typeof node === 'string') {
121
+ const s = node.trim();
122
+ if (s.startsWith('{')) {
123
+ try { node = JSON.parse(s); continue; } catch { /* not JSON — it's the message */ }
124
+ }
125
+ return s.slice(0, 400);
126
+ }
127
+ if (typeof node !== 'object') break;
128
+ if (typeof node.message === 'string') return node.message.slice(0, 400);
129
+ node = node.error ?? node.detail ?? null;
130
+ }
131
+ }
132
+ return raw.slice(0, 400);
133
+ }
134
+
135
+ /**
136
+ * classifyLaunchFailure(result) → null | { kind, httpStatus, message }
137
+ *
138
+ * `result` is parseResultEvent()'s output. Returns null for every run that
139
+ * did real work (or failed for a reason that is not a first-request API
140
+ * rejection) — the narrowness is the point; see the module header.
141
+ * HTTP 429 is excluded: the rate-limit pause path owns it.
142
+ */
143
+ function classifyLaunchFailure(result) {
144
+ if (!result) return null;
145
+ const numTurns = result.numTurns ?? 0;
146
+ const outputTokens = result.outputTokens ?? 0;
147
+ if (numTurns > 1 || outputTokens > 0) return null;
148
+ const text = result.resultText || '';
149
+ const marker = /API Error:?\s*(\d{3})?/i.exec(text);
150
+ if (!marker && !(result.isError && result.apiErrorStatus)) return null;
151
+ const httpStatus = (marker && marker[1] ? Number(marker[1]) : null) ?? result.apiErrorStatus ?? null;
152
+ if (httpStatus === 429) return null;
153
+ const message = extractApiMessage(text.replace(/^.*?API Error:?\s*(\d{3})?\s*/i, '')) || text.slice(0, 400);
154
+ let kind;
155
+ if (httpStatus === 400) {
156
+ kind = /thinking|not supported for this model|output_config|effort/i.test(text)
157
+ ? LAUNCH_FAILURE_KINDS.MODEL_CONFIG_REJECTED
158
+ : LAUNCH_FAILURE_KINDS.BAD_REQUEST;
159
+ } else if (httpStatus === 401 || httpStatus === 403) {
160
+ kind = LAUNCH_FAILURE_KINDS.AUTH_FAILED;
161
+ } else if (httpStatus === 404 && /model/i.test(text)) {
162
+ kind = LAUNCH_FAILURE_KINDS.MODEL_NOT_FOUND;
163
+ } else if ((httpStatus !== null && httpStatus >= 500) || /overloaded/i.test(text)) {
164
+ kind = LAUNCH_FAILURE_KINDS.API_OVERLOADED;
165
+ } else {
166
+ kind = LAUNCH_FAILURE_KINDS.API_ERROR;
167
+ }
168
+ return { kind, httpStatus, message };
169
+ }
170
+
171
+ /** Did this run get at least one real model turn? (The half-open probe's "close the breaker" evidence.) */
172
+ function resultShowsRealTurn(result) {
173
+ if (!result) return false;
174
+ return (result.numTurns ?? 0) > 1 || (result.outputTokens ?? 0) > 0;
175
+ }
176
+
177
+ // ─── Circuit breaker ────────────────────────────────────────────────────────
178
+
179
+ /** After this many consecutive failed probes the block stays open until the CLI version changes or a human resets it. */
180
+ const LAUNCH_BLOCK_MAX_ATTEMPTS = 8;
181
+ /** A probe that has not reported back in this long is presumed dead; the next tick may probe again. */
182
+ const LAUNCH_PROBE_STALE_MS = 30 * 60_000;
183
+
184
+ const BACKOFF_BASE_MS = {
185
+ [LAUNCH_FAILURE_KINDS.MODEL_CONFIG_REJECTED]: 5 * 60_000,
186
+ [LAUNCH_FAILURE_KINDS.BAD_REQUEST]: 5 * 60_000,
187
+ [LAUNCH_FAILURE_KINDS.AUTH_FAILED]: 5 * 60_000,
188
+ [LAUNCH_FAILURE_KINDS.MODEL_NOT_FOUND]: 10 * 60_000,
189
+ [LAUNCH_FAILURE_KINDS.API_OVERLOADED]: 60_000,
190
+ [LAUNCH_FAILURE_KINDS.API_ERROR]: 2 * 60_000,
191
+ };
192
+ const BACKOFF_CAP_MS = 60 * 60_000;
193
+
194
+ /** Exponential backoff for the Nth consecutive failure (attempts ≥ 1), capped at one hour. */
195
+ function backoffMsFor(kind, attempts) {
196
+ const base = BACKOFF_BASE_MS[kind] ?? BACKOFF_BASE_MS[LAUNCH_FAILURE_KINDS.API_ERROR];
197
+ const n = Math.max(0, (attempts ?? 1) - 1);
198
+ return Math.min(BACKOFF_CAP_MS, base * 2 ** n);
199
+ }
200
+
201
+ /**
202
+ * Environment the scheduler applies to a launch as a DEGRADED-MODE
203
+ * mitigation for a kind, or null when there is none. For the issue-#11
204
+ * signature the CLI's `MAX_THINKING_TOKENS=0` switches extended thinking off
205
+ * entirely, so an older CLI stops sending the rejected `thinking` block at
206
+ * all — the run proceeds without thinking rather than not at all. Harmless
207
+ * if the CLI ignores it (the probe simply fails again and the backoff holds).
208
+ */
209
+ function mitigationEnvFor(kind) {
210
+ if (kind === LAUNCH_FAILURE_KINDS.MODEL_CONFIG_REJECTED) return { MAX_THINKING_TOKENS: '0' };
211
+ return null;
212
+ }
213
+
214
+ /** Circuit-breaker key: the launch persona, because `agentType` is what selects the `--model`. */
215
+ function launchBlockKeyFor(job) {
216
+ return (job && typeof job.agentType === 'string' && job.agentType) || 'default';
217
+ }
218
+
219
+ /** Operator-facing explanation + the one action that clears the condition. */
220
+ function launchFailureHint(kind, { claudeVersion, mitigationInForce = false } = {}) {
221
+ const ver = claudeVersion ? `installed Claude CLI ${claudeVersion}` : 'installed Claude CLI';
222
+ switch (kind) {
223
+ case LAUNCH_FAILURE_KINDS.MODEL_CONFIG_REJECTED:
224
+ return mitigationInForce
225
+ ? `The ${ver} sends a thinking parameter this model rejects, and disabling thinking (MAX_THINKING_TOKENS=0) did not get past it. Update the CLI (\`claude update\` or \`npm i -g @anthropic-ai/claude-code@latest\`); the queue resumes automatically when the version changes.`
226
+ : `The ${ver} sends a thinking parameter this model rejects (HTTP 400 on the first request — no work was attempted). Update the CLI (\`claude update\` or \`npm i -g @anthropic-ai/claude-code@latest\`); until then jobs re-probe with thinking disabled, and the queue resumes automatically when the version changes.`;
227
+ case LAUNCH_FAILURE_KINDS.AUTH_FAILED:
228
+ return `The API rejected the CLI's credentials (HTTP 401/403) before any work started. Run \`claude login\` (or check the model is enabled for this account), then press Retry now.`;
229
+ case LAUNCH_FAILURE_KINDS.MODEL_NOT_FOUND:
230
+ return `The pinned --model does not exist for the ${ver} / this account (HTTP 404). Fix the persona's \`model:\` in the Agent Library or update the CLI, then press Retry now.`;
231
+ case LAUNCH_FAILURE_KINDS.API_OVERLOADED:
232
+ return 'The API is overloaded or unavailable (HTTP 5xx/529). Nothing is wrong with the PRD; the scheduler re-probes with backoff and resumes on its own.';
233
+ case LAUNCH_FAILURE_KINDS.BAD_REQUEST:
234
+ return `The API rejected the launch request (HTTP 400) before any work started. Check the ${ver} against the pinned model, then press Retry now.`;
235
+ default:
236
+ return 'The first API request of the run failed before any work started. The scheduler re-probes with backoff; press Retry now to probe immediately.';
237
+ }
238
+ }
239
+
240
+ /**
241
+ * armLaunchBlock(prev, failure) → block
242
+ *
243
+ * Opens (or re-opens with a longer backoff) the breaker for one key after a
244
+ * launch failure. `prev` is the existing block for the key or null; a
245
+ * different `kind` than before restarts the attempt count, the same kind
246
+ * escalates it. Beyond LAUNCH_BLOCK_MAX_ATTEMPTS `until` becomes null:
247
+ * blocked indefinitely — only a CLI version change or a human Retry clears
248
+ * it, and the UI says so.
249
+ */
250
+ function armLaunchBlock(prev, { kind, httpStatus, message, now, claudeVersion, slug, runId, mitigationApplied = false }) {
251
+ const sameKind = prev && prev.kind === kind;
252
+ const attempts = sameKind ? (prev.attempts ?? 0) + 1 : 1;
253
+ const exhausted = attempts >= LAUNCH_BLOCK_MAX_ATTEMPTS;
254
+ const mitigationEnv = mitigationEnvFor(kind);
255
+ return {
256
+ kind,
257
+ httpStatus: httpStatus ?? null,
258
+ message: String(message || '').slice(0, 400),
259
+ hint: launchFailureHint(kind, { claudeVersion, mitigationInForce: mitigationApplied }),
260
+ since: sameKind && prev.since ? prev.since : new Date(now).toISOString(),
261
+ lastAt: new Date(now).toISOString(),
262
+ until: exhausted ? null : new Date(now + backoffMsFor(kind, attempts)).toISOString(),
263
+ attempts,
264
+ exhausted,
265
+ claudeVersion: claudeVersion ?? null,
266
+ lastSlug: slug ?? null,
267
+ lastRunId: runId ?? null,
268
+ mitigationEnv,
269
+ mitigationApplied,
270
+ probing: null,
271
+ };
272
+ }
273
+
274
+ /**
275
+ * evaluateLaunchGate(block, { now, claudeVersion }) →
276
+ * { state: 'open' | 'blocked' | 'probe', reason }
277
+ *
278
+ * 'open' — no block, or the CLI version changed since it was armed (the
279
+ * caller should drop the block: the environment was replaced).
280
+ * 'blocked' — inside the backoff window, exhausted, or a probe is already in
281
+ * flight. `reason` is the row-level hold text.
282
+ * 'probe' — backoff elapsed: let exactly ONE job through as the probe.
283
+ */
284
+ function evaluateLaunchGate(block, { now, claudeVersion } = {}) {
285
+ if (!block) return { state: 'open', reason: null };
286
+ if (claudeVersion && block.claudeVersion && claudeVersion !== block.claudeVersion) {
287
+ return { state: 'open', reason: `cli-version-changed (${block.claudeVersion} → ${claudeVersion})` };
288
+ }
289
+ const t = typeof now === 'number' ? now : Date.now();
290
+ if (block.probing && block.probing.at) {
291
+ const age = t - Date.parse(block.probing.at);
292
+ if (Number.isFinite(age) && age >= 0 && age < LAUNCH_PROBE_STALE_MS) {
293
+ return { state: 'blocked', reason: `launch blocked (${block.kind}) — probe ${block.probing.slug} in flight` };
294
+ }
295
+ }
296
+ if (block.until === null || block.until === undefined) {
297
+ return { state: 'blocked', reason: `launch blocked (${block.kind}) after ${block.attempts} failed probe(s) — ${block.hint}` };
298
+ }
299
+ const until = Date.parse(block.until);
300
+ if (Number.isFinite(until) && t < until) {
301
+ const mins = Math.max(1, Math.round((until - t) / 60_000));
302
+ return { state: 'blocked', reason: `launch blocked (${block.kind}) — re-probe in ${mins} min. ${block.hint}` };
303
+ }
304
+ return { state: 'probe', reason: `launch probe (${block.kind}) — attempt ${(block.attempts ?? 0) + 1}` };
305
+ }
306
+
307
+ /**
308
+ * Terminal-reason taxonomy for a finalized job row (issue #11 list A2). A
309
+ * closed set so operators never have to open a transcript to tell a
310
+ * non-start from an implementation failure from a verifier downgrade.
311
+ */
312
+ function deriveTerminalReason({ effectiveStatus, exitCode, verifyResult, sigtermOverride, worktreeIntegrationFailure }) {
313
+ if (worktreeIntegrationFailure) return 'worktree_integration_failed';
314
+ if (effectiveStatus === 'completed') return 'completed';
315
+ if (sigtermOverride) return 'signal_kill_with_commit';
316
+ if (exitCode === 143 || exitCode === 137) return 'signal_kill';
317
+ if (typeof exitCode === 'number' && exitCode !== 0) return `impl_failed:exit_${exitCode}`;
318
+ if (effectiveStatus === 'needs_review' && verifyResult && verifyResult.verdict) return `verifier:${verifyResult.verdict}`;
319
+ return effectiveStatus || 'unknown';
320
+ }
321
+
322
+ /**
323
+ * Write the per-run `<slug>.outcome.json` sidecar (issue #11 list B5): the
324
+ * handful of numbers that make fleet health computable without parsing
325
+ * transcripts. Best-effort, never throws.
326
+ */
327
+ function writeOutcomeSidecar(runDir, slug, outcome) {
328
+ if (!runDir || !slug) return null;
329
+ const p = path.join(runDir, `${slug}.outcome.json`);
330
+ try {
331
+ const tmp = `${p}.tmp`;
332
+ fs.writeFileSync(tmp, JSON.stringify({ slug, writtenAt: new Date().toISOString(), ...outcome }, null, 2));
333
+ fs.renameSync(tmp, p);
334
+ return p;
335
+ } catch {
336
+ return null;
337
+ }
338
+ }
339
+
340
+ module.exports = {
341
+ LAUNCH_FAILURE_KINDS,
342
+ LAUNCH_BLOCK_MAX_ATTEMPTS,
343
+ LAUNCH_PROBE_STALE_MS,
344
+ parseResultEvent,
345
+ readResultEvent,
346
+ extractApiMessage,
347
+ classifyLaunchFailure,
348
+ resultShowsRealTurn,
349
+ backoffMsFor,
350
+ mitigationEnvFor,
351
+ launchBlockKeyFor,
352
+ launchFailureHint,
353
+ armLaunchBlock,
354
+ evaluateLaunchGate,
355
+ deriveTerminalReason,
356
+ writeOutcomeSidecar,
357
+ };
@@ -0,0 +1,134 @@
1
+ 'use strict';
2
+
3
+ /**
4
+ * loadGate.cjs — CPU-load launch gate for the scheduler (PRD 1085).
5
+ *
6
+ * The scheduler already gates launches on free memory (scheduler.cjs's
7
+ * memoryLimitedBatchSize) and on a static per-project cap
8
+ * (schedulerConfig.projectJobCap). Neither sees CPU. Observed 2026-09-01 on
9
+ * starry-night-ships: 4 concurrent executors each running a Godot test
10
+ * battery under its own Xvfb, loadavg 12.95 on 14 cores. Under that
11
+ * contention every 8-minute battery stretches, executors hit their own
12
+ * `timeout`, the verifier files FAIL/FATAL → needs_review → an auto-fix
13
+ * chain with inflated estimates launches MORE batteries. A static cap cannot
14
+ * see that feedback loop; the 1-minute load average can.
15
+ *
16
+ * Gate ordering at the call site (scheduler.cjs tickQueue):
17
+ * global sessionSlots pool → per-project cap → memory → LOAD (innermost).
18
+ * This is one more predicate inside the existing pick path — never a second
19
+ * pool, and it only WITHHOLDS launches; running jobs are never touched.
20
+ *
21
+ * Everything here is pure and injectable (`loadavg`, `cores`, `now`) so the
22
+ * decision, the audit rate-limit and the escalation are unit-testable without
23
+ * real load or a fake clock hack on `Date`.
24
+ */
25
+
26
+ const os = require('node:os');
27
+ const { execFileSync } = require('node:child_process');
28
+ const { loadGateThreshold, JOB_OVERRUN_FLOOR_MS } = require('./schedulerConfig.cjs');
29
+
30
+ // One audit row per this interval while continuously gated — the poll loop
31
+ // ticks every 60 s and a saturated box stays saturated for a while; auditing
32
+ // every tick would bury the signal in its own noise.
33
+ const AUDIT_INTERVAL_MS = 10 * 60_000;
34
+
35
+ /**
36
+ * isLoadGated(loadavg1, cores, threshold) → boolean
37
+ *
38
+ * True when the 1-minute load average per core exceeds `threshold`. The 5-
39
+ * and 15-minute averages are deliberately NOT consulted: a finished battery
40
+ * should free launches within a minute or two, not a quarter hour. A
41
+ * threshold of 0 (or anything non-positive) disables the gate. Zero/unknown
42
+ * cores never gates (os.cpus() can be empty in some containers).
43
+ */
44
+ function isLoadGated(loadavg1, cores, threshold) {
45
+ if (!(threshold > 0)) return false;
46
+ if (!(cores > 0)) return false;
47
+ if (!Number.isFinite(loadavg1) || loadavg1 <= 0) return false; // [0,0,0] on Windows/unsupported
48
+ return loadavg1 / cores > threshold;
49
+ }
50
+
51
+ /** Best-effort top-N CPU consumers (Linux only). Returns [] anywhere else or on error. */
52
+ function topCpuConsumers(n = 3) {
53
+ if (process.platform !== 'linux') return [];
54
+ try {
55
+ const out = execFileSync('ps', ['-eo', 'pid,pcpu,comm', '--sort=-pcpu'], { encoding: 'utf8', timeout: 2000 });
56
+ return out.split('\n').slice(1, 1 + n).map((l) => l.trim()).filter(Boolean);
57
+ } catch {
58
+ return [];
59
+ }
60
+ }
61
+
62
+ /**
63
+ * createLoadGate(opts) → { evaluate, snapshot }
64
+ *
65
+ * Holds the small amount of state the gate needs across ticks (when it was
66
+ * last audited, when the current gated stretch began). `evaluate({ bypass })`
67
+ * returns the decision for THIS tick:
68
+ * { gated, ratio, threshold, loadavg1, cores, bypassed,
69
+ * shouldAudit, escalate, gatedSinceMs }
70
+ * - shouldAudit: true at most once per AUDIT_INTERVAL_MS while gated.
71
+ * - escalate: true once the current gated stretch exceeds JOB_OVERRUN_FLOOR_MS
72
+ * (45 min by default) — the caller warn-logs with topCpuConsumers().
73
+ * - bypassed: `bypass` was set (an explicit human Run now) and the gate would
74
+ * otherwise have held; the caller launches anyway and logs that it did.
75
+ *
76
+ * `snapshot()` is what buildScheduleStatePayload exposes as `loadGate`.
77
+ */
78
+ function createLoadGate({
79
+ loadavg = () => os.loadavg(),
80
+ cores = () => (os.cpus() || []).length,
81
+ now = () => Date.now(),
82
+ threshold = loadGateThreshold,
83
+ auditIntervalMs = AUDIT_INTERVAL_MS,
84
+ escalateAfterMs = JOB_OVERRUN_FLOOR_MS,
85
+ } = {}) {
86
+ let lastAuditAt = null; // null = never audited; the first gated tick always audits
87
+ let gatedSince = null;
88
+ let last = null;
89
+
90
+ function evaluate({ bypass = false } = {}) {
91
+ const t = now();
92
+ const [l1] = loadavg();
93
+ const c = cores();
94
+ const th = typeof threshold === 'function' ? threshold() : threshold;
95
+ const ratio = c > 0 && Number.isFinite(l1) ? l1 / c : 0;
96
+ const wouldGate = isLoadGated(l1, c, th);
97
+
98
+ if (wouldGate) {
99
+ if (gatedSince === null) gatedSince = t;
100
+ } else {
101
+ gatedSince = null;
102
+ }
103
+ const gated = wouldGate && !bypass;
104
+ const bypassed = wouldGate && bypass;
105
+
106
+ let shouldAudit = false;
107
+ if (gated && (lastAuditAt === null || t - lastAuditAt >= auditIntervalMs)) {
108
+ shouldAudit = true;
109
+ lastAuditAt = t;
110
+ }
111
+ const gatedSinceMs = gatedSince === null ? 0 : t - gatedSince;
112
+ const escalate = gated && gatedSinceMs >= escalateAfterMs;
113
+
114
+ last = {
115
+ gated,
116
+ bypassed,
117
+ ratio: Number(ratio.toFixed(3)),
118
+ threshold: th,
119
+ loadavg1: Number.isFinite(l1) ? Number(l1.toFixed(2)) : null,
120
+ cores: c,
121
+ gatedSinceMs,
122
+ at: new Date(t).toISOString(),
123
+ };
124
+ return { ...last, shouldAudit, escalate };
125
+ }
126
+
127
+ function snapshot() {
128
+ return last;
129
+ }
130
+
131
+ return { evaluate, snapshot };
132
+ }
133
+
134
+ module.exports = { isLoadGated, createLoadGate, topCpuConsumers, AUDIT_INTERVAL_MS };