claude-code-session-manager 0.75.3 → 0.77.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/assets/{AgentLibrary-CzQqcObq.js → AgentLibrary-B2ie8bbw.js} +2 -2
- package/dist/assets/{DataModel-Bj_WlLz8.js → DataModel-BIJPYw32.js} +1 -1
- package/dist/assets/{History-DnSi_OHm.js → History-CeY6dk9S.js} +2 -2
- package/dist/assets/{Hooks-0BB0dp3S.js → Hooks-BFH2ocKg.js} +2 -2
- package/dist/assets/{HostBilko-DHpwwsLQ.js → HostBilko-36gj9wLz.js} +1 -1
- package/dist/assets/{Library-CaJVqVvi.js → Library-C-hBct39.js} +1 -1
- package/dist/assets/{ListDetail-C1W2HmC2.js → ListDetail-CNq64VWV.js} +1 -1
- package/dist/assets/{MarkdownEditor-5Ob9FW3z.js → MarkdownEditor-Bh3qt5-1.js} +1 -1
- package/dist/assets/{McpServers-JxCSfm1S.js → McpServers-DpGN0oyz.js} +1 -1
- package/dist/assets/{Memory-BDeqlqwH.js → Memory-D59hUjC4.js} +6 -6
- package/dist/assets/{Panel-Dh9ZHuEj.js → Panel-DCgbaoci.js} +1 -1
- package/dist/assets/{Permissions-DXy-CbEY.js → Permissions-DAmQ0DYV.js} +2 -2
- package/dist/assets/{Plugins-_n1Iuc8T.js → Plugins-Dyfgn6Is.js} +2 -2
- package/dist/assets/{ProvenanceBadge-BP_evfxE.js → ProvenanceBadge-BiYhPO1U.js} +1 -1
- package/dist/assets/SaveBar-RV7B6sOh.js +1 -0
- package/dist/assets/Scheduler-BPaNqx1b.js +14 -0
- package/dist/assets/{ScopeSwitcher-CAWzM6RI.js → ScopeSwitcher-P4mdLGNU.js} +1 -1
- package/dist/assets/{Settings-DRRozLyT.js → Settings-BL4vf5aX.js} +1 -1
- package/dist/assets/{SkillReferenceGraph-DGHDWlz4.js → SkillReferenceGraph-BRBDyi1_.js} +1 -1
- package/dist/assets/{Skills-D8L66eiX.js → Skills-BV08gDUH.js} +2 -2
- package/dist/assets/{SystemPrompt-CYtUsonD.js → SystemPrompt-CLftSsDw.js} +1 -1
- package/dist/assets/TagLibrary-Bp8jGsd5.js +1 -0
- package/dist/assets/{TiptapBody-B2hRgbPE.js → TiptapBody-jCpuB6E5.js} +1 -1
- package/dist/assets/{Toggle-BTwsbxam.js → Toggle-D2paA1xf.js} +1 -1
- package/dist/assets/{index-DijufvkJ.js → index-BDRSqBl3.js} +704 -704
- package/dist/assets/{index-CMLnzdZC.css → index-CYhdtisq.css} +1 -1
- package/dist/assets/{settingsSchema-D6wzxAi6.js → settingsSchema-6IOLjZZN.js} +1 -1
- package/dist/index.html +2 -2
- package/package.json +8 -2
- package/plugins/session-manager-dev/skills/develop/standards.md +1 -1
- package/scripts/lib/activeSessions.cjs +116 -6
- package/scripts/project-pages-logic/dist/logic.cjs +4709 -0
- package/scripts/render-project-pages/dist/renderer.cjs +18900 -0
- package/scripts/render-project-pages.cjs +70 -0
- package/scripts/scheduler-mcp-server.cjs +269 -96
- package/scripts/validate-project-pages-summary.cjs +62 -0
- package/src/main/__tests__/agentModelResolve.test.cjs +66 -0
- package/src/main/__tests__/epicStatusMirror.test.cjs +110 -0
- package/src/main/__tests__/health-delegation-chain.test.cjs +106 -0
- package/src/main/__tests__/prdAdminRoutes.test.cjs +295 -0
- package/src/main/__tests__/prdAgentType.test.cjs +103 -0
- package/src/main/__tests__/prdCreate.test.cjs +247 -0
- package/src/main/__tests__/prdFrontmatterAgentType.test.cjs +117 -0
- package/src/main/__tests__/prdFrontmatterQuietMachine.test.cjs +108 -0
- package/src/main/__tests__/projectHomeAdminRoutes.test.cjs +485 -0
- package/src/main/__tests__/projectPages.test.cjs +73 -1
- package/src/main/__tests__/rcaReport.test.cjs +54 -0
- package/src/main/__tests__/runVerify.test.cjs +94 -0
- package/src/main/__tests__/scheduler-autofix-select.test.cjs +58 -3
- package/src/main/__tests__/scheduler-bash-timeout-env.test.cjs +103 -0
- package/src/main/__tests__/scheduler-commit-guard-noop.test.cjs +41 -0
- package/src/main/__tests__/scheduler-effective-concurrency.test.cjs +10 -0
- package/src/main/__tests__/scheduler-foreign-wip-manifest.test.cjs +78 -0
- package/src/main/__tests__/scheduler-inplace-salvage.test.cjs +242 -0
- package/src/main/__tests__/scheduler-investigation-prompt.test.cjs +31 -0
- package/src/main/__tests__/scheduler-launch-failure.test.cjs +201 -0
- package/src/main/__tests__/scheduler-leftover-fields.test.cjs +52 -0
- package/src/main/__tests__/scheduler-looks-done.test.cjs +241 -0
- package/src/main/__tests__/scheduler-prd-persona-spawn.test.cjs +135 -0
- package/src/main/__tests__/scheduler-quiet-machine-lease.test.cjs +222 -0
- package/src/main/__tests__/scheduler-reap-dead-running-jobs.test.cjs +207 -1
- package/src/main/__tests__/scheduler-shared-tree-guard.test.cjs +212 -0
- package/src/main/__tests__/scheduler-stranded-investigation.test.cjs +185 -0
- package/src/main/__tests__/scheduler-worktree-cap-defer.test.cjs +194 -0
- package/src/main/__tests__/seedAgentPersonas.test.cjs +75 -14
- package/src/main/__tests__/seedSchedulerMcp.test.cjs +66 -0
- package/src/main/__tests__/uniquePrdNumbers.test.cjs +14 -5
- package/src/main/bilkoHost.cjs +4 -3
- package/src/main/chatRunner.cjs +6 -1
- package/src/main/config.cjs +25 -33
- package/src/main/health.cjs +153 -2
- package/src/main/index.cjs +64 -5
- package/src/main/ipcSchemas.cjs +69 -1
- package/src/main/lib/__tests__/activeIndexRebuild.test.cjs +179 -0
- package/src/main/lib/__tests__/childWithLog.test.cjs +141 -0
- package/src/main/lib/__tests__/delegationReadiness.test.cjs +391 -42
- package/src/main/lib/__tests__/ephemeralCwd.test.cjs +91 -0
- package/src/main/lib/__tests__/epicWorktreeMint.test.cjs +5 -3
- package/src/main/lib/__tests__/fixChainDepth.test.cjs +40 -0
- package/src/main/lib/__tests__/gitWorktree.test.cjs +290 -5
- package/src/main/lib/__tests__/gitWorktreeSalvage.test.cjs +107 -0
- package/src/main/lib/__tests__/gitWorktreeSalvageDelta.test.cjs +153 -0
- package/src/main/lib/__tests__/jobWorktree.test.cjs +6 -4
- package/src/main/lib/__tests__/landedSinceRun.test.cjs +73 -0
- package/src/main/lib/__tests__/launchFailure.test.cjs +220 -0
- package/src/main/lib/__tests__/loadGate.test.cjs +159 -0
- package/src/main/lib/__tests__/mcpToolCatalog.test.cjs +102 -0
- package/src/main/lib/__tests__/opsOwnership.test.cjs +7 -0
- package/src/main/lib/__tests__/opsRootAbsoluteCwd.test.cjs +151 -0
- package/src/main/lib/__tests__/opsRootResolve.test.cjs +149 -0
- package/src/main/lib/__tests__/prdDeclaredPaths.test.cjs +82 -0
- package/src/main/lib/__tests__/projectRootResolve.test.cjs +148 -0
- package/src/main/lib/__tests__/queueHealth.test.cjs +58 -0
- package/src/main/lib/__tests__/quietMachineLease.test.cjs +39 -0
- package/src/main/lib/__tests__/reaperHelpers.test.cjs +133 -0
- package/src/main/lib/__tests__/schedulerBatchDepends.test.cjs +19 -9
- package/src/main/lib/__tests__/schedulerBatchFairness.test.cjs +213 -0
- package/src/main/lib/__tests__/schedulerBatchLaunchHold.test.cjs +125 -0
- package/src/main/lib/__tests__/schedulerBatchProjectCap.test.cjs +127 -0
- package/src/main/lib/__tests__/schedulerBatchQuietMachine.test.cjs +109 -0
- package/src/main/lib/__tests__/schedulerMcpServerHeadlessRefusal.test.cjs +71 -0
- package/src/main/lib/__tests__/schedulerMcpServerHelp.test.cjs +217 -0
- package/src/main/lib/__tests__/schedulerMcpServerProjectHome.test.cjs +350 -0
- package/src/main/lib/activeIndexMerge.cjs +15 -0
- package/src/main/lib/activeIndexRebuild.cjs +133 -0
- package/src/main/lib/agentModelResolve.cjs +58 -0
- package/src/main/lib/buildTarget.cjs +3 -2
- package/src/main/lib/childWithLog.cjs +69 -2
- package/src/main/lib/claudeBin.cjs +54 -1
- package/src/main/lib/crossProjectFeedback.cjs +8 -1
- package/src/main/lib/definitionOfDone.cjs +3 -2
- package/src/main/lib/delegationReadiness.cjs +514 -26
- package/src/main/lib/ephemeralCwd.cjs +78 -0
- package/src/main/lib/epicDelegationStats.cjs +2 -1
- package/src/main/lib/epicMint.cjs +17 -1
- package/src/main/lib/epicStatusMirror.cjs +95 -0
- package/src/main/lib/epicValidationHook.cjs +2 -1
- package/src/main/lib/epicWorktreeMint.cjs +5 -2
- package/src/main/lib/fixChainDepth.cjs +45 -0
- package/src/main/lib/gitWorktree.cjs +520 -21
- package/src/main/lib/jobWorktree.cjs +2 -0
- package/src/main/lib/landedSinceRun.cjs +55 -0
- package/src/main/lib/launchFailure.cjs +357 -0
- package/src/main/lib/loadGate.cjs +134 -0
- package/src/main/lib/mcpToolCatalog.cjs +370 -0
- package/src/main/lib/opsErrorLog.cjs +12 -1
- package/src/main/lib/opsOwnership.cjs +106 -0
- package/src/main/lib/prdAdminRoutes.cjs +43 -3
- package/src/main/lib/prdAgentType.cjs +84 -0
- package/src/main/lib/prdCreate.cjs +103 -15
- package/src/main/lib/prdDeclaredPaths.cjs +70 -0
- package/src/main/lib/prdFrontmatter.cjs +17 -3
- package/src/main/lib/prdLocations.cjs +13 -6
- package/src/main/lib/projectHomeAdminRoutes.cjs +402 -0
- package/src/main/lib/projectPageSummarySchema.cjs +181 -0
- package/src/main/lib/projectRootResolve.cjs +134 -0
- package/src/main/lib/promptSessionSchema.cjs +7 -0
- package/src/main/lib/queueHealth.cjs +38 -0
- package/src/main/lib/queueStore.cjs +40 -7
- package/src/main/lib/quietMachineLease.cjs +48 -0
- package/src/main/lib/rcaReport.cjs +54 -4
- package/src/main/lib/reaperHelpers.cjs +64 -1
- package/src/main/lib/scheduleJobSchema.cjs +31 -0
- package/src/main/lib/scheduleJobTransitions.cjs +6 -2
- package/src/main/lib/schedulerBatch.cjs +301 -55
- package/src/main/lib/schedulerConfig.cjs +99 -0
- package/src/main/projectBrief.cjs +3 -2
- package/src/main/projectPages.cjs +162 -3
- package/src/main/promptSessionTranscript.cjs +0 -0
- package/src/main/pty.cjs +5 -0
- package/src/main/queueOps.cjs +15 -8
- package/src/main/runVerify.cjs +50 -9
- package/src/main/scheduler/prdParser.cjs +18 -1
- package/src/main/scheduler.cjs +1701 -130
- package/src/main/seedAgentPersonas.cjs +62 -21
- package/src/main/seedSchedulerMcp.cjs +58 -4
- package/src/main/templates/project-pages-catalog.json +741 -0
- package/src/main/templates/project-pages-pipeline.md +417 -0
- package/src/preload/api.d.ts +187 -3
- package/src/preload/index.cjs +9 -0
- package/src/seed/agents/project-home-builder.md +59 -0
- package/dist/assets/SaveBar-D-gCUx4n.js +0 -1
- package/dist/assets/Scheduler-Bpd4OGju.js +0 -14
- package/dist/assets/TagLibrary-E5CLeuVk.js +0 -1
|
@@ -62,6 +62,8 @@ module.exports = {
|
|
|
62
62
|
createJobWorktree: gitWorktree.createJobWorktree,
|
|
63
63
|
integrateJobBranch: gitWorktree.integrateJobBranch,
|
|
64
64
|
cleanupJobWorktree: gitWorktree.cleanupJobWorktree,
|
|
65
|
+
salvageJobWorktreeDiff: gitWorktree.salvageJobWorktreeDiff,
|
|
66
|
+
salvageJobDirtyDelta: gitWorktree.salvageJobDirtyDelta,
|
|
65
67
|
parseWorktreeListPorcelain: gitWorktree.parseWorktreeListPorcelain,
|
|
66
68
|
reconcileWorktreesOnBoot: (cwds) => gitWorktree.reconcileWorktreesOnBoot(cwds, { kind: KIND }),
|
|
67
69
|
// Test-only escape hatch for the in-memory concurrency counter.
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* landedSinceRun.cjs — widened, path-scoped commit evidence for the reverify
|
|
5
|
+
* self-heal pass (PRD 1102).
|
|
6
|
+
*
|
|
7
|
+
* committedInWindow (scheduler.cjs) only sees commits inside
|
|
8
|
+
* [startedAt, finishedAt+60s] — a commit that lands later (a retry, a
|
|
9
|
+
* sibling run, a human) is invisible to it. landedSinceRun has no upper
|
|
10
|
+
* bound, but narrows the OTHER way that committedInWindow is dangerously
|
|
11
|
+
* broad: it is scoped to paths the PRD itself declares, so an unrelated
|
|
12
|
+
* commit elsewhere in the repo is not credited to this job (see
|
|
13
|
+
* scheduler.cjs's healRefusalReason for why repo-wide, unscoped evidence is
|
|
14
|
+
* not attribution).
|
|
15
|
+
*
|
|
16
|
+
* Pure git wrapper — no fetch, no scheduler state. Callers that need remote
|
|
17
|
+
* commits visible (e.g. a job that committed in a since-removed worktree)
|
|
18
|
+
* must call scheduler.cjs's fetchAllRefs(cwd) first, same as
|
|
19
|
+
* committedInWindow's own callers do.
|
|
20
|
+
*/
|
|
21
|
+
|
|
22
|
+
const { execFile } = require('node:child_process');
|
|
23
|
+
|
|
24
|
+
const LANDED_SINCE_RUN_TIMEOUT_MS = 10_000;
|
|
25
|
+
|
|
26
|
+
/**
|
|
27
|
+
* Commits in `cwd` since `sinceIso` (no upper bound) that touch any of
|
|
28
|
+
* `paths`. Never throws — git-unavailable, a non-repo cwd, or an empty
|
|
29
|
+
* `paths` list all resolve to `[]` rather than fabricating evidence.
|
|
30
|
+
*
|
|
31
|
+
* @param {string} cwd
|
|
32
|
+
* @param {string} sinceIso
|
|
33
|
+
* @param {string[]} paths
|
|
34
|
+
* @param {{ timeoutMs?: number }} [opts]
|
|
35
|
+
* @returns {Promise<string[]>} full commit SHAs, newest first
|
|
36
|
+
*/
|
|
37
|
+
function landedSinceRun(cwd, sinceIso, paths, { timeoutMs = LANDED_SINCE_RUN_TIMEOUT_MS } = {}) {
|
|
38
|
+
return new Promise((resolve) => {
|
|
39
|
+
if (!cwd || !sinceIso || !Array.isArray(paths) || paths.length === 0) {
|
|
40
|
+
resolve([]);
|
|
41
|
+
return;
|
|
42
|
+
}
|
|
43
|
+
execFile(
|
|
44
|
+
'git',
|
|
45
|
+
['-C', cwd, 'log', '--all', `--since=${sinceIso}`, '--format=%H', '--', ...paths],
|
|
46
|
+
{ timeout: timeoutMs, windowsHide: true },
|
|
47
|
+
(err, stdout) => {
|
|
48
|
+
if (err) { resolve([]); return; }
|
|
49
|
+
resolve(String(stdout || '').trim().split('\n').filter(Boolean));
|
|
50
|
+
},
|
|
51
|
+
);
|
|
52
|
+
});
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
module.exports = { landedSinceRun, LANDED_SINCE_RUN_TIMEOUT_MS };
|
|
@@ -0,0 +1,357 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* launchFailure.cjs — detects a headless `claude -p` run that NEVER RAN, and
|
|
5
|
+
* models the per-persona launch circuit breaker the scheduler routes on.
|
|
6
|
+
*
|
|
7
|
+
* Incident (GitHub issue #11, 2026-09-02, macOS): the installed Claude CLI
|
|
8
|
+
* sent `thinking.type.enabled` to a model that only accepts
|
|
9
|
+
* `thinking.type.adaptive`. The API answered HTTP 400 on the very first
|
|
10
|
+
* request, so every job did zero turns, spent zero output tokens, wrote zero
|
|
11
|
+
* files, exited 1 in ~25 s — and was recorded as `failed` with `error: null`,
|
|
12
|
+
* indistinguishable from a real implementation failure. The auto-fix
|
|
13
|
+
* investigation probe then launched with the same CLI and died the same way.
|
|
14
|
+
* 12 of 41 transcripts in one project over a month were this signature.
|
|
15
|
+
*
|
|
16
|
+
* Two facts this module makes first-class:
|
|
17
|
+
*
|
|
18
|
+
* 1. A NON-RUN is not a failure of the PRD. `classifyLaunchFailure` is
|
|
19
|
+
* deliberately narrow — it only fires when the transcript's `result`
|
|
20
|
+
* event shows no real turn (num_turns ≤ 1 AND output_tokens = 0) AND the
|
|
21
|
+
* result text carries the CLI's literal `API Error:` prefix. Anything
|
|
22
|
+
* that did a turn, or failed without the API marker, is somebody else's
|
|
23
|
+
* classification (rate-limit, network, transient, verifier).
|
|
24
|
+
*
|
|
25
|
+
* 2. The environment is broken, not the job — so the scheduler must stop
|
|
26
|
+
* re-dispatching identical launches (each one is a wasted 25 s + a
|
|
27
|
+
* misleading `failed` row + a doomed investigation) while still
|
|
28
|
+
* self-healing the moment the environment is fixed. That is a circuit
|
|
29
|
+
* breaker keyed by the launch persona (`agentType` → model): closed
|
|
30
|
+
* (normal) → open (blocked, exponential backoff) → half-open (exactly one
|
|
31
|
+
* probe job goes through) → closed again on a real turn. A CLI version
|
|
32
|
+
* change (the actual fix for the incident: `claude update`) short-circuits
|
|
33
|
+
* the backoff so the queue resumes on the next tick, not the next hour.
|
|
34
|
+
*
|
|
35
|
+
* Pure and Electron-free: every function here takes plain values (a parsed
|
|
36
|
+
* result event, a block record, `now`) so the whole state machine is
|
|
37
|
+
* unit-testable without a spawn. scheduler.cjs owns the I/O around it.
|
|
38
|
+
*/
|
|
39
|
+
|
|
40
|
+
const fs = require('node:fs');
|
|
41
|
+
const path = require('node:path');
|
|
42
|
+
const { readTail } = require('./fileTail.cjs');
|
|
43
|
+
|
|
44
|
+
const LAUNCH_FAILURE_KINDS = Object.freeze({
|
|
45
|
+
/** HTTP 400 naming a thinking/effort/config parameter the model rejects — the issue-#11 signature. */
|
|
46
|
+
MODEL_CONFIG_REJECTED: 'model_config_rejected',
|
|
47
|
+
/** Any other HTTP 400 on the first request (malformed request body, unsupported flag combo). */
|
|
48
|
+
BAD_REQUEST: 'bad_request',
|
|
49
|
+
/** HTTP 401/403 — the CLI's credentials are missing, expired, or lack access to the model. */
|
|
50
|
+
AUTH_FAILED: 'auth_failed',
|
|
51
|
+
/** HTTP 404 that names the model — the pinned `--model` does not exist for this account/CLI. */
|
|
52
|
+
MODEL_NOT_FOUND: 'model_not_found',
|
|
53
|
+
/** HTTP 5xx / 529 / "Overloaded" — the API itself is unavailable right now. */
|
|
54
|
+
API_OVERLOADED: 'api_overloaded',
|
|
55
|
+
/** Any other first-request API error. */
|
|
56
|
+
API_ERROR: 'api_error',
|
|
57
|
+
});
|
|
58
|
+
|
|
59
|
+
/** Tail bytes scanned for the `result` event — same budget classifyRunOutcome uses. */
|
|
60
|
+
const RESULT_TAIL_BYTES = 65536;
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* Parse the LAST `{"type":"result",...}` stream-json event out of a log tail.
|
|
64
|
+
* Returns a flat, typed summary or null when no result event is present
|
|
65
|
+
* (the process died before the harness could emit one — that is
|
|
66
|
+
* reaperHelpers' `no_result`, not a launch failure).
|
|
67
|
+
*/
|
|
68
|
+
function parseResultEvent(text) {
|
|
69
|
+
if (!text) return null;
|
|
70
|
+
let last = null;
|
|
71
|
+
for (const line of String(text).split('\n')) {
|
|
72
|
+
const t = line.trim();
|
|
73
|
+
if (!t.startsWith('{') || !t.includes('"type":"result"')) continue;
|
|
74
|
+
try {
|
|
75
|
+
const obj = JSON.parse(t);
|
|
76
|
+
if (obj && obj.type === 'result') last = obj;
|
|
77
|
+
} catch { /* partial line at the tail boundary */ }
|
|
78
|
+
}
|
|
79
|
+
if (!last) return null;
|
|
80
|
+
const usage = last.usage && typeof last.usage === 'object' ? last.usage : {};
|
|
81
|
+
const num = (v) => (typeof v === 'number' && Number.isFinite(v) ? v : null);
|
|
82
|
+
return {
|
|
83
|
+
subtype: typeof last.subtype === 'string' ? last.subtype : '',
|
|
84
|
+
isError: last.is_error === true,
|
|
85
|
+
numTurns: num(last.num_turns),
|
|
86
|
+
outputTokens: num(usage.output_tokens),
|
|
87
|
+
inputTokens: num(usage.input_tokens),
|
|
88
|
+
apiErrorStatus: num(last.api_error_status),
|
|
89
|
+
totalCostUsd: num(last.total_cost_usd),
|
|
90
|
+
durationMs: num(last.duration_ms),
|
|
91
|
+
terminalReason: typeof last.terminal_reason === 'string' ? last.terminal_reason : null,
|
|
92
|
+
resultText: typeof last.result === 'string' ? last.result : '',
|
|
93
|
+
};
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
function readResultEvent(logPath) {
|
|
97
|
+
try {
|
|
98
|
+
return parseResultEvent(readTail(logPath, RESULT_TAIL_BYTES));
|
|
99
|
+
} catch {
|
|
100
|
+
return null;
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
/**
|
|
105
|
+
* Pull the human-readable message out of the CLI's `API Error: <status> <json>`
|
|
106
|
+
* text. The body nests unpredictably (`{"error":{"message":...}}`, or
|
|
107
|
+
* `{"detail":{"error":"<json string with message>"}}` as in issue #11), so
|
|
108
|
+
* this walks any `message`/`error` chain it can parse and falls back to the
|
|
109
|
+
* raw text, bounded.
|
|
110
|
+
*/
|
|
111
|
+
function extractApiMessage(text) {
|
|
112
|
+
const raw = String(text || '').trim();
|
|
113
|
+
const jsonStart = raw.indexOf('{');
|
|
114
|
+
if (jsonStart >= 0) {
|
|
115
|
+
let node;
|
|
116
|
+
try { node = JSON.parse(raw.slice(jsonStart)); } catch { node = null; }
|
|
117
|
+
let depth = 0;
|
|
118
|
+
while (node && depth < 6) {
|
|
119
|
+
depth += 1;
|
|
120
|
+
if (typeof node === 'string') {
|
|
121
|
+
const s = node.trim();
|
|
122
|
+
if (s.startsWith('{')) {
|
|
123
|
+
try { node = JSON.parse(s); continue; } catch { /* not JSON — it's the message */ }
|
|
124
|
+
}
|
|
125
|
+
return s.slice(0, 400);
|
|
126
|
+
}
|
|
127
|
+
if (typeof node !== 'object') break;
|
|
128
|
+
if (typeof node.message === 'string') return node.message.slice(0, 400);
|
|
129
|
+
node = node.error ?? node.detail ?? null;
|
|
130
|
+
}
|
|
131
|
+
}
|
|
132
|
+
return raw.slice(0, 400);
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/**
|
|
136
|
+
* classifyLaunchFailure(result) → null | { kind, httpStatus, message }
|
|
137
|
+
*
|
|
138
|
+
* `result` is parseResultEvent()'s output. Returns null for every run that
|
|
139
|
+
* did real work (or failed for a reason that is not a first-request API
|
|
140
|
+
* rejection) — the narrowness is the point; see the module header.
|
|
141
|
+
* HTTP 429 is excluded: the rate-limit pause path owns it.
|
|
142
|
+
*/
|
|
143
|
+
function classifyLaunchFailure(result) {
|
|
144
|
+
if (!result) return null;
|
|
145
|
+
const numTurns = result.numTurns ?? 0;
|
|
146
|
+
const outputTokens = result.outputTokens ?? 0;
|
|
147
|
+
if (numTurns > 1 || outputTokens > 0) return null;
|
|
148
|
+
const text = result.resultText || '';
|
|
149
|
+
const marker = /API Error:?\s*(\d{3})?/i.exec(text);
|
|
150
|
+
if (!marker && !(result.isError && result.apiErrorStatus)) return null;
|
|
151
|
+
const httpStatus = (marker && marker[1] ? Number(marker[1]) : null) ?? result.apiErrorStatus ?? null;
|
|
152
|
+
if (httpStatus === 429) return null;
|
|
153
|
+
const message = extractApiMessage(text.replace(/^.*?API Error:?\s*(\d{3})?\s*/i, '')) || text.slice(0, 400);
|
|
154
|
+
let kind;
|
|
155
|
+
if (httpStatus === 400) {
|
|
156
|
+
kind = /thinking|not supported for this model|output_config|effort/i.test(text)
|
|
157
|
+
? LAUNCH_FAILURE_KINDS.MODEL_CONFIG_REJECTED
|
|
158
|
+
: LAUNCH_FAILURE_KINDS.BAD_REQUEST;
|
|
159
|
+
} else if (httpStatus === 401 || httpStatus === 403) {
|
|
160
|
+
kind = LAUNCH_FAILURE_KINDS.AUTH_FAILED;
|
|
161
|
+
} else if (httpStatus === 404 && /model/i.test(text)) {
|
|
162
|
+
kind = LAUNCH_FAILURE_KINDS.MODEL_NOT_FOUND;
|
|
163
|
+
} else if ((httpStatus !== null && httpStatus >= 500) || /overloaded/i.test(text)) {
|
|
164
|
+
kind = LAUNCH_FAILURE_KINDS.API_OVERLOADED;
|
|
165
|
+
} else {
|
|
166
|
+
kind = LAUNCH_FAILURE_KINDS.API_ERROR;
|
|
167
|
+
}
|
|
168
|
+
return { kind, httpStatus, message };
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
/** Did this run get at least one real model turn? (The half-open probe's "close the breaker" evidence.) */
|
|
172
|
+
function resultShowsRealTurn(result) {
|
|
173
|
+
if (!result) return false;
|
|
174
|
+
return (result.numTurns ?? 0) > 1 || (result.outputTokens ?? 0) > 0;
|
|
175
|
+
}
|
|
176
|
+
|
|
177
|
+
// ─── Circuit breaker ────────────────────────────────────────────────────────
|
|
178
|
+
|
|
179
|
+
/** After this many consecutive failed probes the block stays open until the CLI version changes or a human resets it. */
|
|
180
|
+
const LAUNCH_BLOCK_MAX_ATTEMPTS = 8;
|
|
181
|
+
/** A probe that has not reported back in this long is presumed dead; the next tick may probe again. */
|
|
182
|
+
const LAUNCH_PROBE_STALE_MS = 30 * 60_000;
|
|
183
|
+
|
|
184
|
+
const BACKOFF_BASE_MS = {
|
|
185
|
+
[LAUNCH_FAILURE_KINDS.MODEL_CONFIG_REJECTED]: 5 * 60_000,
|
|
186
|
+
[LAUNCH_FAILURE_KINDS.BAD_REQUEST]: 5 * 60_000,
|
|
187
|
+
[LAUNCH_FAILURE_KINDS.AUTH_FAILED]: 5 * 60_000,
|
|
188
|
+
[LAUNCH_FAILURE_KINDS.MODEL_NOT_FOUND]: 10 * 60_000,
|
|
189
|
+
[LAUNCH_FAILURE_KINDS.API_OVERLOADED]: 60_000,
|
|
190
|
+
[LAUNCH_FAILURE_KINDS.API_ERROR]: 2 * 60_000,
|
|
191
|
+
};
|
|
192
|
+
const BACKOFF_CAP_MS = 60 * 60_000;
|
|
193
|
+
|
|
194
|
+
/** Exponential backoff for the Nth consecutive failure (attempts ≥ 1), capped at one hour. */
|
|
195
|
+
function backoffMsFor(kind, attempts) {
|
|
196
|
+
const base = BACKOFF_BASE_MS[kind] ?? BACKOFF_BASE_MS[LAUNCH_FAILURE_KINDS.API_ERROR];
|
|
197
|
+
const n = Math.max(0, (attempts ?? 1) - 1);
|
|
198
|
+
return Math.min(BACKOFF_CAP_MS, base * 2 ** n);
|
|
199
|
+
}
|
|
200
|
+
|
|
201
|
+
/**
|
|
202
|
+
* Environment the scheduler applies to a launch as a DEGRADED-MODE
|
|
203
|
+
* mitigation for a kind, or null when there is none. For the issue-#11
|
|
204
|
+
* signature the CLI's `MAX_THINKING_TOKENS=0` switches extended thinking off
|
|
205
|
+
* entirely, so an older CLI stops sending the rejected `thinking` block at
|
|
206
|
+
* all — the run proceeds without thinking rather than not at all. Harmless
|
|
207
|
+
* if the CLI ignores it (the probe simply fails again and the backoff holds).
|
|
208
|
+
*/
|
|
209
|
+
function mitigationEnvFor(kind) {
|
|
210
|
+
if (kind === LAUNCH_FAILURE_KINDS.MODEL_CONFIG_REJECTED) return { MAX_THINKING_TOKENS: '0' };
|
|
211
|
+
return null;
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
/** Circuit-breaker key: the launch persona, because `agentType` is what selects the `--model`. */
|
|
215
|
+
function launchBlockKeyFor(job) {
|
|
216
|
+
return (job && typeof job.agentType === 'string' && job.agentType) || 'default';
|
|
217
|
+
}
|
|
218
|
+
|
|
219
|
+
/** Operator-facing explanation + the one action that clears the condition. */
|
|
220
|
+
function launchFailureHint(kind, { claudeVersion, mitigationInForce = false } = {}) {
|
|
221
|
+
const ver = claudeVersion ? `installed Claude CLI ${claudeVersion}` : 'installed Claude CLI';
|
|
222
|
+
switch (kind) {
|
|
223
|
+
case LAUNCH_FAILURE_KINDS.MODEL_CONFIG_REJECTED:
|
|
224
|
+
return mitigationInForce
|
|
225
|
+
? `The ${ver} sends a thinking parameter this model rejects, and disabling thinking (MAX_THINKING_TOKENS=0) did not get past it. Update the CLI (\`claude update\` or \`npm i -g @anthropic-ai/claude-code@latest\`); the queue resumes automatically when the version changes.`
|
|
226
|
+
: `The ${ver} sends a thinking parameter this model rejects (HTTP 400 on the first request — no work was attempted). Update the CLI (\`claude update\` or \`npm i -g @anthropic-ai/claude-code@latest\`); until then jobs re-probe with thinking disabled, and the queue resumes automatically when the version changes.`;
|
|
227
|
+
case LAUNCH_FAILURE_KINDS.AUTH_FAILED:
|
|
228
|
+
return `The API rejected the CLI's credentials (HTTP 401/403) before any work started. Run \`claude login\` (or check the model is enabled for this account), then press Retry now.`;
|
|
229
|
+
case LAUNCH_FAILURE_KINDS.MODEL_NOT_FOUND:
|
|
230
|
+
return `The pinned --model does not exist for the ${ver} / this account (HTTP 404). Fix the persona's \`model:\` in the Agent Library or update the CLI, then press Retry now.`;
|
|
231
|
+
case LAUNCH_FAILURE_KINDS.API_OVERLOADED:
|
|
232
|
+
return 'The API is overloaded or unavailable (HTTP 5xx/529). Nothing is wrong with the PRD; the scheduler re-probes with backoff and resumes on its own.';
|
|
233
|
+
case LAUNCH_FAILURE_KINDS.BAD_REQUEST:
|
|
234
|
+
return `The API rejected the launch request (HTTP 400) before any work started. Check the ${ver} against the pinned model, then press Retry now.`;
|
|
235
|
+
default:
|
|
236
|
+
return 'The first API request of the run failed before any work started. The scheduler re-probes with backoff; press Retry now to probe immediately.';
|
|
237
|
+
}
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
/**
|
|
241
|
+
* armLaunchBlock(prev, failure) → block
|
|
242
|
+
*
|
|
243
|
+
* Opens (or re-opens with a longer backoff) the breaker for one key after a
|
|
244
|
+
* launch failure. `prev` is the existing block for the key or null; a
|
|
245
|
+
* different `kind` than before restarts the attempt count, the same kind
|
|
246
|
+
* escalates it. Beyond LAUNCH_BLOCK_MAX_ATTEMPTS `until` becomes null:
|
|
247
|
+
* blocked indefinitely — only a CLI version change or a human Retry clears
|
|
248
|
+
* it, and the UI says so.
|
|
249
|
+
*/
|
|
250
|
+
function armLaunchBlock(prev, { kind, httpStatus, message, now, claudeVersion, slug, runId, mitigationApplied = false }) {
|
|
251
|
+
const sameKind = prev && prev.kind === kind;
|
|
252
|
+
const attempts = sameKind ? (prev.attempts ?? 0) + 1 : 1;
|
|
253
|
+
const exhausted = attempts >= LAUNCH_BLOCK_MAX_ATTEMPTS;
|
|
254
|
+
const mitigationEnv = mitigationEnvFor(kind);
|
|
255
|
+
return {
|
|
256
|
+
kind,
|
|
257
|
+
httpStatus: httpStatus ?? null,
|
|
258
|
+
message: String(message || '').slice(0, 400),
|
|
259
|
+
hint: launchFailureHint(kind, { claudeVersion, mitigationInForce: mitigationApplied }),
|
|
260
|
+
since: sameKind && prev.since ? prev.since : new Date(now).toISOString(),
|
|
261
|
+
lastAt: new Date(now).toISOString(),
|
|
262
|
+
until: exhausted ? null : new Date(now + backoffMsFor(kind, attempts)).toISOString(),
|
|
263
|
+
attempts,
|
|
264
|
+
exhausted,
|
|
265
|
+
claudeVersion: claudeVersion ?? null,
|
|
266
|
+
lastSlug: slug ?? null,
|
|
267
|
+
lastRunId: runId ?? null,
|
|
268
|
+
mitigationEnv,
|
|
269
|
+
mitigationApplied,
|
|
270
|
+
probing: null,
|
|
271
|
+
};
|
|
272
|
+
}
|
|
273
|
+
|
|
274
|
+
/**
|
|
275
|
+
* evaluateLaunchGate(block, { now, claudeVersion }) →
|
|
276
|
+
* { state: 'open' | 'blocked' | 'probe', reason }
|
|
277
|
+
*
|
|
278
|
+
* 'open' — no block, or the CLI version changed since it was armed (the
|
|
279
|
+
* caller should drop the block: the environment was replaced).
|
|
280
|
+
* 'blocked' — inside the backoff window, exhausted, or a probe is already in
|
|
281
|
+
* flight. `reason` is the row-level hold text.
|
|
282
|
+
* 'probe' — backoff elapsed: let exactly ONE job through as the probe.
|
|
283
|
+
*/
|
|
284
|
+
function evaluateLaunchGate(block, { now, claudeVersion } = {}) {
|
|
285
|
+
if (!block) return { state: 'open', reason: null };
|
|
286
|
+
if (claudeVersion && block.claudeVersion && claudeVersion !== block.claudeVersion) {
|
|
287
|
+
return { state: 'open', reason: `cli-version-changed (${block.claudeVersion} → ${claudeVersion})` };
|
|
288
|
+
}
|
|
289
|
+
const t = typeof now === 'number' ? now : Date.now();
|
|
290
|
+
if (block.probing && block.probing.at) {
|
|
291
|
+
const age = t - Date.parse(block.probing.at);
|
|
292
|
+
if (Number.isFinite(age) && age >= 0 && age < LAUNCH_PROBE_STALE_MS) {
|
|
293
|
+
return { state: 'blocked', reason: `launch blocked (${block.kind}) — probe ${block.probing.slug} in flight` };
|
|
294
|
+
}
|
|
295
|
+
}
|
|
296
|
+
if (block.until === null || block.until === undefined) {
|
|
297
|
+
return { state: 'blocked', reason: `launch blocked (${block.kind}) after ${block.attempts} failed probe(s) — ${block.hint}` };
|
|
298
|
+
}
|
|
299
|
+
const until = Date.parse(block.until);
|
|
300
|
+
if (Number.isFinite(until) && t < until) {
|
|
301
|
+
const mins = Math.max(1, Math.round((until - t) / 60_000));
|
|
302
|
+
return { state: 'blocked', reason: `launch blocked (${block.kind}) — re-probe in ${mins} min. ${block.hint}` };
|
|
303
|
+
}
|
|
304
|
+
return { state: 'probe', reason: `launch probe (${block.kind}) — attempt ${(block.attempts ?? 0) + 1}` };
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
/**
|
|
308
|
+
* Terminal-reason taxonomy for a finalized job row (issue #11 list A2). A
|
|
309
|
+
* closed set so operators never have to open a transcript to tell a
|
|
310
|
+
* non-start from an implementation failure from a verifier downgrade.
|
|
311
|
+
*/
|
|
312
|
+
function deriveTerminalReason({ effectiveStatus, exitCode, verifyResult, sigtermOverride, worktreeIntegrationFailure }) {
|
|
313
|
+
if (worktreeIntegrationFailure) return 'worktree_integration_failed';
|
|
314
|
+
if (effectiveStatus === 'completed') return 'completed';
|
|
315
|
+
if (sigtermOverride) return 'signal_kill_with_commit';
|
|
316
|
+
if (exitCode === 143 || exitCode === 137) return 'signal_kill';
|
|
317
|
+
if (typeof exitCode === 'number' && exitCode !== 0) return `impl_failed:exit_${exitCode}`;
|
|
318
|
+
if (effectiveStatus === 'needs_review' && verifyResult && verifyResult.verdict) return `verifier:${verifyResult.verdict}`;
|
|
319
|
+
return effectiveStatus || 'unknown';
|
|
320
|
+
}
|
|
321
|
+
|
|
322
|
+
/**
|
|
323
|
+
* Write the per-run `<slug>.outcome.json` sidecar (issue #11 list B5): the
|
|
324
|
+
* handful of numbers that make fleet health computable without parsing
|
|
325
|
+
* transcripts. Best-effort, never throws.
|
|
326
|
+
*/
|
|
327
|
+
function writeOutcomeSidecar(runDir, slug, outcome) {
|
|
328
|
+
if (!runDir || !slug) return null;
|
|
329
|
+
const p = path.join(runDir, `${slug}.outcome.json`);
|
|
330
|
+
try {
|
|
331
|
+
const tmp = `${p}.tmp`;
|
|
332
|
+
fs.writeFileSync(tmp, JSON.stringify({ slug, writtenAt: new Date().toISOString(), ...outcome }, null, 2));
|
|
333
|
+
fs.renameSync(tmp, p);
|
|
334
|
+
return p;
|
|
335
|
+
} catch {
|
|
336
|
+
return null;
|
|
337
|
+
}
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
module.exports = {
|
|
341
|
+
LAUNCH_FAILURE_KINDS,
|
|
342
|
+
LAUNCH_BLOCK_MAX_ATTEMPTS,
|
|
343
|
+
LAUNCH_PROBE_STALE_MS,
|
|
344
|
+
parseResultEvent,
|
|
345
|
+
readResultEvent,
|
|
346
|
+
extractApiMessage,
|
|
347
|
+
classifyLaunchFailure,
|
|
348
|
+
resultShowsRealTurn,
|
|
349
|
+
backoffMsFor,
|
|
350
|
+
mitigationEnvFor,
|
|
351
|
+
launchBlockKeyFor,
|
|
352
|
+
launchFailureHint,
|
|
353
|
+
armLaunchBlock,
|
|
354
|
+
evaluateLaunchGate,
|
|
355
|
+
deriveTerminalReason,
|
|
356
|
+
writeOutcomeSidecar,
|
|
357
|
+
};
|
|
@@ -0,0 +1,134 @@
|
|
|
1
|
+
'use strict';
|
|
2
|
+
|
|
3
|
+
/**
|
|
4
|
+
* loadGate.cjs — CPU-load launch gate for the scheduler (PRD 1085).
|
|
5
|
+
*
|
|
6
|
+
* The scheduler already gates launches on free memory (scheduler.cjs's
|
|
7
|
+
* memoryLimitedBatchSize) and on a static per-project cap
|
|
8
|
+
* (schedulerConfig.projectJobCap). Neither sees CPU. Observed 2026-09-01 on
|
|
9
|
+
* starry-night-ships: 4 concurrent executors each running a Godot test
|
|
10
|
+
* battery under its own Xvfb, loadavg 12.95 on 14 cores. Under that
|
|
11
|
+
* contention every 8-minute battery stretches, executors hit their own
|
|
12
|
+
* `timeout`, the verifier files FAIL/FATAL → needs_review → an auto-fix
|
|
13
|
+
* chain with inflated estimates launches MORE batteries. A static cap cannot
|
|
14
|
+
* see that feedback loop; the 1-minute load average can.
|
|
15
|
+
*
|
|
16
|
+
* Gate ordering at the call site (scheduler.cjs tickQueue):
|
|
17
|
+
* global sessionSlots pool → per-project cap → memory → LOAD (innermost).
|
|
18
|
+
* This is one more predicate inside the existing pick path — never a second
|
|
19
|
+
* pool, and it only WITHHOLDS launches; running jobs are never touched.
|
|
20
|
+
*
|
|
21
|
+
* Everything here is pure and injectable (`loadavg`, `cores`, `now`) so the
|
|
22
|
+
* decision, the audit rate-limit and the escalation are unit-testable without
|
|
23
|
+
* real load or a fake clock hack on `Date`.
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
const os = require('node:os');
|
|
27
|
+
const { execFileSync } = require('node:child_process');
|
|
28
|
+
const { loadGateThreshold, JOB_OVERRUN_FLOOR_MS } = require('./schedulerConfig.cjs');
|
|
29
|
+
|
|
30
|
+
// One audit row per this interval while continuously gated — the poll loop
|
|
31
|
+
// ticks every 60 s and a saturated box stays saturated for a while; auditing
|
|
32
|
+
// every tick would bury the signal in its own noise.
|
|
33
|
+
const AUDIT_INTERVAL_MS = 10 * 60_000;
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* isLoadGated(loadavg1, cores, threshold) → boolean
|
|
37
|
+
*
|
|
38
|
+
* True when the 1-minute load average per core exceeds `threshold`. The 5-
|
|
39
|
+
* and 15-minute averages are deliberately NOT consulted: a finished battery
|
|
40
|
+
* should free launches within a minute or two, not a quarter hour. A
|
|
41
|
+
* threshold of 0 (or anything non-positive) disables the gate. Zero/unknown
|
|
42
|
+
* cores never gates (os.cpus() can be empty in some containers).
|
|
43
|
+
*/
|
|
44
|
+
function isLoadGated(loadavg1, cores, threshold) {
|
|
45
|
+
if (!(threshold > 0)) return false;
|
|
46
|
+
if (!(cores > 0)) return false;
|
|
47
|
+
if (!Number.isFinite(loadavg1) || loadavg1 <= 0) return false; // [0,0,0] on Windows/unsupported
|
|
48
|
+
return loadavg1 / cores > threshold;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
/** Best-effort top-N CPU consumers (Linux only). Returns [] anywhere else or on error. */
|
|
52
|
+
function topCpuConsumers(n = 3) {
|
|
53
|
+
if (process.platform !== 'linux') return [];
|
|
54
|
+
try {
|
|
55
|
+
const out = execFileSync('ps', ['-eo', 'pid,pcpu,comm', '--sort=-pcpu'], { encoding: 'utf8', timeout: 2000 });
|
|
56
|
+
return out.split('\n').slice(1, 1 + n).map((l) => l.trim()).filter(Boolean);
|
|
57
|
+
} catch {
|
|
58
|
+
return [];
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
/**
|
|
63
|
+
* createLoadGate(opts) → { evaluate, snapshot }
|
|
64
|
+
*
|
|
65
|
+
* Holds the small amount of state the gate needs across ticks (when it was
|
|
66
|
+
* last audited, when the current gated stretch began). `evaluate({ bypass })`
|
|
67
|
+
* returns the decision for THIS tick:
|
|
68
|
+
* { gated, ratio, threshold, loadavg1, cores, bypassed,
|
|
69
|
+
* shouldAudit, escalate, gatedSinceMs }
|
|
70
|
+
* - shouldAudit: true at most once per AUDIT_INTERVAL_MS while gated.
|
|
71
|
+
* - escalate: true once the current gated stretch exceeds JOB_OVERRUN_FLOOR_MS
|
|
72
|
+
* (45 min by default) — the caller warn-logs with topCpuConsumers().
|
|
73
|
+
* - bypassed: `bypass` was set (an explicit human Run now) and the gate would
|
|
74
|
+
* otherwise have held; the caller launches anyway and logs that it did.
|
|
75
|
+
*
|
|
76
|
+
* `snapshot()` is what buildScheduleStatePayload exposes as `loadGate`.
|
|
77
|
+
*/
|
|
78
|
+
function createLoadGate({
|
|
79
|
+
loadavg = () => os.loadavg(),
|
|
80
|
+
cores = () => (os.cpus() || []).length,
|
|
81
|
+
now = () => Date.now(),
|
|
82
|
+
threshold = loadGateThreshold,
|
|
83
|
+
auditIntervalMs = AUDIT_INTERVAL_MS,
|
|
84
|
+
escalateAfterMs = JOB_OVERRUN_FLOOR_MS,
|
|
85
|
+
} = {}) {
|
|
86
|
+
let lastAuditAt = null; // null = never audited; the first gated tick always audits
|
|
87
|
+
let gatedSince = null;
|
|
88
|
+
let last = null;
|
|
89
|
+
|
|
90
|
+
function evaluate({ bypass = false } = {}) {
|
|
91
|
+
const t = now();
|
|
92
|
+
const [l1] = loadavg();
|
|
93
|
+
const c = cores();
|
|
94
|
+
const th = typeof threshold === 'function' ? threshold() : threshold;
|
|
95
|
+
const ratio = c > 0 && Number.isFinite(l1) ? l1 / c : 0;
|
|
96
|
+
const wouldGate = isLoadGated(l1, c, th);
|
|
97
|
+
|
|
98
|
+
if (wouldGate) {
|
|
99
|
+
if (gatedSince === null) gatedSince = t;
|
|
100
|
+
} else {
|
|
101
|
+
gatedSince = null;
|
|
102
|
+
}
|
|
103
|
+
const gated = wouldGate && !bypass;
|
|
104
|
+
const bypassed = wouldGate && bypass;
|
|
105
|
+
|
|
106
|
+
let shouldAudit = false;
|
|
107
|
+
if (gated && (lastAuditAt === null || t - lastAuditAt >= auditIntervalMs)) {
|
|
108
|
+
shouldAudit = true;
|
|
109
|
+
lastAuditAt = t;
|
|
110
|
+
}
|
|
111
|
+
const gatedSinceMs = gatedSince === null ? 0 : t - gatedSince;
|
|
112
|
+
const escalate = gated && gatedSinceMs >= escalateAfterMs;
|
|
113
|
+
|
|
114
|
+
last = {
|
|
115
|
+
gated,
|
|
116
|
+
bypassed,
|
|
117
|
+
ratio: Number(ratio.toFixed(3)),
|
|
118
|
+
threshold: th,
|
|
119
|
+
loadavg1: Number.isFinite(l1) ? Number(l1.toFixed(2)) : null,
|
|
120
|
+
cores: c,
|
|
121
|
+
gatedSinceMs,
|
|
122
|
+
at: new Date(t).toISOString(),
|
|
123
|
+
};
|
|
124
|
+
return { ...last, shouldAudit, escalate };
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
function snapshot() {
|
|
128
|
+
return last;
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
return { evaluate, snapshot };
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
module.exports = { isLoadGated, createLoadGate, topCpuConsumers, AUDIT_INTERVAL_MS };
|