@agentproto/apps 0.18.0 → 0.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +2 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.mjs +61 -2
- package/dist/index.mjs.map +1 -1
- package/dist/review-panel/panel.d.ts +1 -1
- package/dist/review-panel/panel.d.ts.map +1 -1
- package/dist/review-panel/panel.generated.d.ts +1 -1
- package/dist/review-panel/panel.generated.d.ts.map +1 -1
- package/dist/review-panel/panel.mjs +1 -1
- package/dist/review-panel/panel.mjs.map +1 -1
- package/dist/review-panel.mjs +1 -1
- package/dist/review-panel.mjs.map +1 -1
- package/dist/store/index.d.ts +61 -0
- package/dist/store/index.d.ts.map +1 -0
- package/dist/store/panel.d.ts +53 -0
- package/dist/store/panel.d.ts.map +1 -0
- package/dist/store/panel.generated.d.ts +13 -0
- package/dist/store/panel.generated.d.ts.map +1 -0
- package/dist/store/panel.mjs +25 -0
- package/dist/store/panel.mjs.map +1 -0
- package/dist/store.mjs +71 -0
- package/dist/store.mjs.map +1 -0
- package/package.json +14 -2
- package/session-steward/.agentproto/APP.md +2 -0
- package/session-steward/.agentproto/workflows/session-steward/WORKFLOW.md +218 -11
- package/session-steward/.agentproto/workflows/session-steward/cron-rules.mjs +526 -0
- package/session-steward/.agentproto/workflows/session-steward/entry.mjs +539 -64
- package/session-steward/.agentproto/workflows/session-steward/origin-policy.mjs +115 -0
- package/session-steward/README.md +19 -2
- package/session-steward/routines/session-steward-hourly/ROUTINE.md +24 -1
- package/session-steward/scripts/sessions-snapshot.sh +85 -0
- package/session-steward/skill/SKILL.md +74 -0
|
@@ -0,0 +1,526 @@
|
|
|
1
|
+
// Mechanical session-health rules, ported from the `kill-idle-sessions`
|
|
2
|
+
// cron prototype (`.plans/session-steward-cron/SESSIONS-LOG.md`, rules 1-9 +
|
|
3
|
+
// the three "Retours steward" dogfood blocks). Each rule here is a PURE
|
|
4
|
+
// function over data the workflow already gathers — `session_list`
|
|
5
|
+
// descriptors, `tool_calls_list` records, `host_load`, `app_state` events —
|
|
6
|
+
// with no I/O, no clock read, no daemon state, so a unit test pins it
|
|
7
|
+
// exactly. `entry.mjs` (the WORKFLOW.md step graph) wires them; the LLM/Jev
|
|
8
|
+
// judge only sees what stays ambiguous after these rules have spoken.
|
|
9
|
+
//
|
|
10
|
+
// Safety shape (mirrors origin-policy.mjs):
|
|
11
|
+
// - a `looping` session is NEVER closed — the proposed action is a nudge
|
|
12
|
+
// interrupt, and only ever a PROPOSAL in the report;
|
|
13
|
+
// - a stall proposes a "continue" nudge, never a close;
|
|
14
|
+
// - at most ONE nudge proposal per session per pass;
|
|
15
|
+
// - the proposed actions are reported in dry run exactly as in apply —
|
|
16
|
+
// `apply` never changes what these pure functions decide, only whether
|
|
17
|
+
// the report is executed elsewhere.
|
|
18
|
+
//
|
|
19
|
+
// Provenance lives in origin-policy.mjs and still wins: a user-origin
|
|
20
|
+
// session is reported as observed, never nudged (a human is in the loop).
|
|
21
|
+
|
|
22
|
+
// ── loop detection (mission item 1) ──────────────────────────────────────
|
|
23
|
+
|
|
24
|
+
/** Rolling window the loop signals are computed over. */
|
|
25
|
+
export const LOOP_WINDOW_MS = 10 * 60 * 1000
|
|
26
|
+
/** Same call signature (verbatim argv) this many times in the window. */
|
|
27
|
+
export const LOOP_VERBATIM_MIN = 3
|
|
28
|
+
/** Distinct/total below this, with at least {@link LOOP_RATIO_MIN_CALLS}
|
|
29
|
+
* calls, is a repetition loop. */
|
|
30
|
+
export const LOOP_RATIO_MAX = 0.2
|
|
31
|
+
export const LOOP_RATIO_MIN_CALLS = 8
|
|
32
|
+
/** The same file re-read this many times in the window. */
|
|
33
|
+
export const LOOP_REREAD_MIN = 4
|
|
34
|
+
|
|
35
|
+
const isNonEmptyString = (v) => typeof v === "string" && v.trim().length > 0
|
|
36
|
+
const str = (v) => (isNonEmptyString(v) ? v.trim() : undefined)
|
|
37
|
+
|
|
38
|
+
/** The verbatim argv signature of one `tool_calls_list` record — tool name +
|
|
39
|
+
* command + args, whitespace-collapsed. This is the "même commande relancée
|
|
40
|
+
* verbatim" signal from the log (a `rg … | head -10` repeated six times). */
|
|
41
|
+
export function callSignature(record) {
|
|
42
|
+
const tool = str(record?.tool) ?? "unknown"
|
|
43
|
+
const command = str(record?.command)
|
|
44
|
+
const args = Array.isArray(record?.args) ? record.args.map(String) : []
|
|
45
|
+
const tail = args.length > 0 ? args.join(" ") : ""
|
|
46
|
+
return `${tool} ${command ?? ""} ${tail}`.replace(/\s+/g, " ").trim()
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
/**
|
|
50
|
+
* Command families the log explicitly calls USEFUL loops, not stuck ones:
|
|
51
|
+
* `watch`, a test/type-check re-run, `git status` between steps, and a
|
|
52
|
+
* `gh pr view`/`checks` watch poll. Their repeats are excluded from the
|
|
53
|
+
* repetition signals, so a watchdog polling PR state for ten minutes is not
|
|
54
|
+
* flagged as a loop (SESSIONS-LOG: "watch polling legitime", "tests re-run").
|
|
55
|
+
*/
|
|
56
|
+
export function isUsefulLoopCommand(signature) {
|
|
57
|
+
const s = String(signature ?? "").toLowerCase()
|
|
58
|
+
if (/(^|[^a-z])watch([^a-z]|$)/.test(s)) return true
|
|
59
|
+
if (/(^|[^a-z])git\s+status([^a-z]|$)/.test(s)) return true
|
|
60
|
+
if (/(pnpm|npm|yarn|npx|bun)\s+(run\s+)?(test|vitest|jest|mocha|check-types|typecheck|build|gate)\b/.test(s)) return true
|
|
61
|
+
if (/(^|[^a-z])(vitest|jest|mocha)([^a-z]|$)/.test(s)) return true
|
|
62
|
+
if (/(^|[^a-z])gh\s+pr\s+(view|checks|status|list|watch)\b/.test(s)) return true
|
|
63
|
+
return false
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
const READ_TOOLS = new Set(["read", "cat", "rg", "grep", "sed", "head", "tail", "less", "view", "readfile"])
|
|
67
|
+
const READ_COMMAND = /\b(cat|rg|grep|sed|head|tail|less|view)\b/
|
|
68
|
+
|
|
69
|
+
/**
|
|
70
|
+
* The file a read-like call targeted, or `null`. Best-effort: an in-agent
|
|
71
|
+
* `Read` call records no shell command (see tool-call-record.ts), so this
|
|
72
|
+
* only resolves for shell-shaped reads (`cat`, `rg`, `sed`, …). Absence is
|
|
73
|
+
* not an error — it just means this call contributes no read-target signal.
|
|
74
|
+
*/
|
|
75
|
+
export function readTargetOf(record) {
|
|
76
|
+
const tool = String(record?.tool ?? "").toLowerCase()
|
|
77
|
+
const command = str(record?.command)
|
|
78
|
+
const args = Array.isArray(record?.args) ? record.args.map(String) : []
|
|
79
|
+
const isRead = READ_TOOLS.has(tool) || (command !== undefined && READ_COMMAND.test(command))
|
|
80
|
+
if (!isRead) return null
|
|
81
|
+
const tokens = command !== undefined ? command.split(/\s+/) : args
|
|
82
|
+
for (const raw of tokens) {
|
|
83
|
+
const t = raw.replace(/^['"]|['"]$/g, "")
|
|
84
|
+
if (!t || t.startsWith("-") || t.includes("=")) continue
|
|
85
|
+
if (t.includes("/") || /\.(md|ts|tsx|js|mjs|cjs|json|txt|py|go|rs|sql|yml|yaml|toml)$/.test(t)) return t
|
|
86
|
+
}
|
|
87
|
+
return args[0] ?? null
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
function topEntry(counts) {
|
|
91
|
+
let best = null
|
|
92
|
+
for (const [key, n] of counts) if (best === null || n > best.count) best = { key, count: n }
|
|
93
|
+
return best
|
|
94
|
+
}
|
|
95
|
+
|
|
96
|
+
/**
|
|
97
|
+
* Detect a repetition loop from one session's recent tool calls. `looping`
|
|
98
|
+
* when, inside `windowMs`:
|
|
99
|
+
* - the same call signature appears verbatim ≥ `verbatimMin`; OR
|
|
100
|
+
* - distinct/total < `ratioMax` over at least `ratioMinCalls` calls; OR
|
|
101
|
+
* - the same file is read ≥ `rereadMin` times.
|
|
102
|
+
* Useful-loop command families are excluded from the counts first (see
|
|
103
|
+
* {@link isUsefulLoopCommand}), so a watchdog poll or a test re-run never
|
|
104
|
+
* trips it. Returns the stats either way, so the report can show its work.
|
|
105
|
+
*/
|
|
106
|
+
export function detectLoop(records, opts = {}) {
|
|
107
|
+
const nowMs = opts.nowMs ?? 0
|
|
108
|
+
const windowMs = opts.windowMs ?? LOOP_WINDOW_MS
|
|
109
|
+
const verbatimMin = opts.verbatimMin ?? LOOP_VERBATIM_MIN
|
|
110
|
+
const ratioMax = opts.ratioMax ?? LOOP_RATIO_MAX
|
|
111
|
+
const ratioMinCalls = opts.ratioMinCalls ?? LOOP_RATIO_MIN_CALLS
|
|
112
|
+
const rereadMin = opts.rereadMin ?? LOOP_REREAD_MIN
|
|
113
|
+
|
|
114
|
+
const inWindow = (Array.isArray(records) ? records : []).filter((r) => {
|
|
115
|
+
const ts = Date.parse(r?.ts)
|
|
116
|
+
return Number.isFinite(ts) && ts <= nowMs + 1000 && nowMs - ts <= windowMs
|
|
117
|
+
})
|
|
118
|
+
const calls = inWindow.filter((r) => !isUsefulLoopCommand(callSignature(r)))
|
|
119
|
+
const counts = new Map()
|
|
120
|
+
const reads = new Map()
|
|
121
|
+
for (const r of calls) {
|
|
122
|
+
const sig = callSignature(r)
|
|
123
|
+
counts.set(sig, (counts.get(sig) ?? 0) + 1)
|
|
124
|
+
const target = readTargetOf(r)
|
|
125
|
+
if (target) reads.set(target, (reads.get(target) ?? 0) + 1)
|
|
126
|
+
}
|
|
127
|
+
const total = calls.length
|
|
128
|
+
const distinct = counts.size
|
|
129
|
+
const ratio = total > 0 ? distinct / total : 1
|
|
130
|
+
const maxVerbatim = topEntry(counts)?.count ?? 0
|
|
131
|
+
const maxReads = topEntry(reads)?.count ?? 0
|
|
132
|
+
const top = topEntry(counts)
|
|
133
|
+
|
|
134
|
+
const reasons = []
|
|
135
|
+
if (maxVerbatim >= verbatimMin) reasons.push(`same call verbatim x${maxVerbatim} in ${Math.round(windowMs / 60_000)}m`)
|
|
136
|
+
if (total >= ratioMinCalls && ratio < ratioMax) reasons.push(`distinct/total ${distinct}/${total} < ${ratioMax}`)
|
|
137
|
+
if (maxReads >= rereadMin) reasons.push(`same file read x${maxReads}`)
|
|
138
|
+
|
|
139
|
+
return {
|
|
140
|
+
looping: reasons.length > 0,
|
|
141
|
+
reasons,
|
|
142
|
+
stats: {
|
|
143
|
+
windowMs,
|
|
144
|
+
total,
|
|
145
|
+
distinct,
|
|
146
|
+
ratio: Math.round(ratio * 100) / 100,
|
|
147
|
+
maxVerbatim,
|
|
148
|
+
maxReads,
|
|
149
|
+
usefulExcluded: inWindow.length - total,
|
|
150
|
+
topCommand: top ? top.key : null,
|
|
151
|
+
topCommandCount: top ? top.count : 0,
|
|
152
|
+
},
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
// ── stall (mission item 2) ───────────────────────────────────────────────
|
|
157
|
+
|
|
158
|
+
/** Busy this many minutes with no new activity is a stall. */
|
|
159
|
+
export const STALL_BUSY_MINUTES = 20
|
|
160
|
+
/** A turn error younger than this, on an idle process, is a stall. */
|
|
161
|
+
export const STALL_ERROR_RECENT_MINUTES = 30
|
|
162
|
+
|
|
163
|
+
/**
|
|
164
|
+
* A stall is either (a) busy longer than {@link STALL_BUSY_MINUTES} with no
|
|
165
|
+
* new activity, or (b) a recent `lastTurnErroredAt` on a process that is NOT
|
|
166
|
+
* busy (the turn already ended in an error; a "continue" restarts it). The
|
|
167
|
+
* proposed action is a single "continue" nudge — never a close.
|
|
168
|
+
*/
|
|
169
|
+
export function detectStall(input = {}) {
|
|
170
|
+
const nowMs = input.nowMs ?? 0
|
|
171
|
+
const busyMinutes = input.busyStallMinutes ?? STALL_BUSY_MINUTES
|
|
172
|
+
const errorRecentMinutes = input.errorRecentMinutes ?? STALL_ERROR_RECENT_MINUTES
|
|
173
|
+
const idle = typeof input.idleMinutes === "number" && Number.isFinite(input.idleMinutes) ? input.idleMinutes : undefined
|
|
174
|
+
if (input.busy === true && idle !== undefined && idle > busyMinutes) {
|
|
175
|
+
return { stalled: true, kind: "busy", reason: `busy ${Math.round(idle)}min > ${busyMinutes}min without new activity` }
|
|
176
|
+
}
|
|
177
|
+
const errTs = input.lastTurnErroredAt ? Date.parse(input.lastTurnErroredAt) : Number.NaN
|
|
178
|
+
if (Number.isFinite(errTs)) {
|
|
179
|
+
const ageMin = (nowMs - errTs) / 60_000
|
|
180
|
+
if (ageMin >= 0 && ageMin <= errorRecentMinutes && input.busy !== true) {
|
|
181
|
+
return { stalled: true, kind: "error", reason: `turn errored ${Math.round(ageMin)}min ago and process idle` }
|
|
182
|
+
}
|
|
183
|
+
}
|
|
184
|
+
return { stalled: false, kind: null, reason: null }
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
// ── never-ran (mission item 3) ───────────────────────────────────────────
|
|
188
|
+
|
|
189
|
+
/**
|
|
190
|
+
* A session that never actually ran: explicit 0 in AND 0 out. `undefined`
|
|
191
|
+
* tokens (not reported) is NOT "never ran" — only a hard 0/0 is. Such a
|
|
192
|
+
* session is `stuck` immediately, without a judge, whatever its idle.
|
|
193
|
+
*/
|
|
194
|
+
export function isNeverRan(session) {
|
|
195
|
+
return session?.tokensIn === 0 && session?.tokensOut === 0
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
// ── fast-path done (mission item 4) ──────────────────────────────────────
|
|
199
|
+
|
|
200
|
+
/** A tool-call record whose tool name is a `message_parent` call. */
|
|
201
|
+
export function isMessageParentCall(record) {
|
|
202
|
+
const tool = String(record?.tool ?? "").toLowerCase()
|
|
203
|
+
return tool.includes("message_parent")
|
|
204
|
+
}
|
|
205
|
+
|
|
206
|
+
/** A `message_parent` call carrying `kind: "done"` in any of the shapes a
|
|
207
|
+
* record can expose it (tool name, command, args, or an explicit `kind`). */
|
|
208
|
+
export function isDoneMessageParent(record) {
|
|
209
|
+
if (!isMessageParentCall(record)) return false
|
|
210
|
+
const blob = `${record?.kind ?? ""} ${record?.command ?? ""} ${Array.isArray(record?.args) ? record.args.join(" ") : ""}`.toLowerCase()
|
|
211
|
+
return /(^|[^a-z])done([^a-z]|$)/.test(blob) || record?.kind === "done"
|
|
212
|
+
}
|
|
213
|
+
|
|
214
|
+
/** A call that commits or opens a PR — the second half of the done fast-path. */
|
|
215
|
+
export function isCommitOrPrCall(record) {
|
|
216
|
+
if (record?.createdPrUrl || typeof record?.createdPrNumber === "number") return true
|
|
217
|
+
const tool = String(record?.tool ?? "").toLowerCase()
|
|
218
|
+
if (tool.includes("git_commit") || tool.includes("create_pull_request") || tool.includes("pr_create")) return true
|
|
219
|
+
const command = String(record?.command ?? "").toLowerCase()
|
|
220
|
+
return /\bgit\s+commit\b/.test(command) || /\bgh\s+pr\s+create\b/.test(command)
|
|
221
|
+
}
|
|
222
|
+
|
|
223
|
+
/**
|
|
224
|
+
* The done fast-path: the last tool call is `message_parent(kind:done)` AND
|
|
225
|
+
* the window shows a commit/PR (or the worktree/PR state or the derived
|
|
226
|
+
* outcome already proves one). That is `done` without spending a judge.
|
|
227
|
+
* Anything short of both halves returns `{ done: false }`.
|
|
228
|
+
*/
|
|
229
|
+
export function detectFastPathDone(input = {}) {
|
|
230
|
+
const calls = Array.isArray(input.toolCalls) ? input.toolCalls : []
|
|
231
|
+
const last = input.lastToolCall ?? (calls.length > 0 ? calls[calls.length - 1] : null)
|
|
232
|
+
const doneMessage = isDoneMessageParent(last) || calls.some(isDoneMessageParent)
|
|
233
|
+
if (!doneMessage) return { done: false, reason: "no message_parent(kind:done)" }
|
|
234
|
+
const prState = input.worktree?.pr?.state
|
|
235
|
+
const merged = prState === "merged" || prState === "MERGED"
|
|
236
|
+
const opened = prState === "open" || prState === "OPEN"
|
|
237
|
+
const outcomePrs = Array.isArray(input.outcome?.pullRequests) ? input.outcome.pullRequests.length : 0
|
|
238
|
+
const hasCommitOrPr = calls.some(isCommitOrPrCall) || merged || opened || outcomePrs > 0
|
|
239
|
+
if (!hasCommitOrPr) return { done: false, reason: "message_parent(kind:done) but no commit/PR" }
|
|
240
|
+
return { done: true, reason: merged || outcomePrs > 0 ? "message_parent(kind:done) + merged/opened PR" : "message_parent(kind:done) + commit" }
|
|
241
|
+
}
|
|
242
|
+
|
|
243
|
+
// ── terminal sessions without an outcome (mission item 5) ────────────────
|
|
244
|
+
|
|
245
|
+
const TERMINAL_STATUSES = new Set(["killed", "exited", "error", "stopped", "completed", "failed"])
|
|
246
|
+
|
|
247
|
+
/**
|
|
248
|
+
* A terminal session that still carries no derived outcome and no wrapup
|
|
249
|
+
* flag is a relabel CANDIDATE — visible instead of invisible, as the log
|
|
250
|
+
* asks. The proposed verdict is `done` when the worktree/PR proves the work
|
|
251
|
+
* merged, else `abandoned` (the ambiguous ones still go to a judge).
|
|
252
|
+
*/
|
|
253
|
+
export function terminalRelabelCandidate(session) {
|
|
254
|
+
if (!TERMINAL_STATUSES.has(String(session?.status ?? ""))) return { candidate: false, reason: "not terminal" }
|
|
255
|
+
if (session?.outcome?.verdict) return { candidate: false, reason: "outcome already recorded" }
|
|
256
|
+
if (session?.wrapupFlag) return { candidate: false, reason: "already flagged" }
|
|
257
|
+
const prState = session?.worktree?.pr?.state
|
|
258
|
+
const merged = prState === "merged" || prState === "MERGED"
|
|
259
|
+
const outcomePrs = Array.isArray(session?.outcome?.pullRequests) ? session.outcome.pullRequests : []
|
|
260
|
+
const mergedPr = merged || outcomePrs.some((p) => p?.state === "MERGED" || p?.state === "merged")
|
|
261
|
+
if (mergedPr) return { candidate: true, proposedVerdict: "done", reason: "terminal, PR merged, no outcome recorded" }
|
|
262
|
+
return { candidate: true, proposedVerdict: "abandoned", reason: "terminal, no outcome recorded" }
|
|
263
|
+
}
|
|
264
|
+
|
|
265
|
+
// ── self-exclusion (mission item 7) ──────────────────────────────────────
|
|
266
|
+
|
|
267
|
+
/** Two origins name the same cron job (`cron:<jobId>`). */
|
|
268
|
+
export function sameCronJob(origin, otherOrigin) {
|
|
269
|
+
return isNonEmptyString(origin) && origin === otherOrigin && origin.startsWith("cron:")
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
/**
|
|
273
|
+
* Whether a candidate is the steward's OWN cron lineage and must never be
|
|
274
|
+
* judged as user work: the caller itself, or an older run of the same cron
|
|
275
|
+
* job (`origin === callerOrigin`, both `cron:<jobId>`). A finished cron run
|
|
276
|
+
* closes itself, or is a certain close — never a judge candidate.
|
|
277
|
+
*/
|
|
278
|
+
export function isSelfExcluded(session, opts = {}) {
|
|
279
|
+
if (!session) return { excluded: true, reason: "no session" }
|
|
280
|
+
if (opts.callerSessionId && session.id === opts.callerSessionId) return { excluded: true, reason: "caller session" }
|
|
281
|
+
if (sameCronJob(session.origin, opts.callerOrigin)) return { excluded: true, reason: "same cron job as caller" }
|
|
282
|
+
return { excluded: false, reason: null }
|
|
283
|
+
}
|
|
284
|
+
|
|
285
|
+
// ── re-check at apply time (mission item 6) ──────────────────────────────
|
|
286
|
+
|
|
287
|
+
/**
|
|
288
|
+
* Re-read a candidate's live state right before acting. If it became busy,
|
|
289
|
+
* or is no longer running/starting, SKIP — the snapshot it was planned from
|
|
290
|
+
* is stale. `live` is a fresh `session_list` row (or `{status,busy}`).
|
|
291
|
+
*/
|
|
292
|
+
export function recheckApply(_entry, live) {
|
|
293
|
+
if (!live) return { proceed: false, reason: "session disappeared before apply" }
|
|
294
|
+
if (live.busy === true) return { proceed: false, reason: "became busy before apply" }
|
|
295
|
+
if (live.status !== "running" && live.status !== "starting") {
|
|
296
|
+
return { proceed: false, reason: `terminal before apply (${live.status})` }
|
|
297
|
+
}
|
|
298
|
+
return { proceed: true, reason: null }
|
|
299
|
+
}
|
|
300
|
+
|
|
301
|
+
// ── verdict memory (mission item 10) ─────────────────────────────────────
|
|
302
|
+
|
|
303
|
+
/** A short, stable fingerprint of the decision-relevant evidence — NOT the
|
|
304
|
+
* transcript text, which changes on every turn. Two passes with the same
|
|
305
|
+
* fingerprint reached the same verdict on unchanged evidence. */
|
|
306
|
+
export function evidenceFingerprint(evidence) {
|
|
307
|
+
const pick = {
|
|
308
|
+
status: evidence?.status ?? null,
|
|
309
|
+
busy: evidence?.busy === true,
|
|
310
|
+
awaitingInput: evidence?.awaitingInput === true,
|
|
311
|
+
keepAlive: evidence?.keepAlive === true,
|
|
312
|
+
turnsCompleted: evidence?.turnsCompleted ?? null,
|
|
313
|
+
tokensIn: evidence?.tokensIn ?? null,
|
|
314
|
+
tokensOut: evidence?.tokensOut ?? null,
|
|
315
|
+
lastToolCall: evidence?.lastToolCall?.tool ?? null,
|
|
316
|
+
toolTotal: evidence?.toolStats?.total ?? null,
|
|
317
|
+
toolDistinct: evidence?.toolStats?.distinct ?? null,
|
|
318
|
+
pr: evidence?.worktree?.pr?.state ?? null,
|
|
319
|
+
lastTurnError: evidence?.lastTurnError ?? null,
|
|
320
|
+
liveChildren: evidence?.liveChildren ?? null,
|
|
321
|
+
continuedTo: evidence?.continuedTo ?? null,
|
|
322
|
+
outcome: evidence?.outcome?.verdict ?? null,
|
|
323
|
+
}
|
|
324
|
+
return JSON.stringify(pick)
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
/**
|
|
328
|
+
* Fold `app_state` `note` events of payload `kind:"steward-verdict"` into a
|
|
329
|
+
* per-session memory: the last verdict, its fingerprint, and how many
|
|
330
|
+
* consecutive passes agreed on the same (verdict, fingerprint). `streak`
|
|
331
|
+
* resets the moment either changes.
|
|
332
|
+
*/
|
|
333
|
+
export function foldVerdictMemory(events) {
|
|
334
|
+
const memory = new Map()
|
|
335
|
+
for (const e of Array.isArray(events) ? events : []) {
|
|
336
|
+
const p = e?.payload
|
|
337
|
+
if (e?.kind !== "note" || p?.kind !== "steward-verdict" || !isNonEmptyString(p.sessionId)) continue
|
|
338
|
+
const prev = memory.get(p.sessionId)
|
|
339
|
+
const streak = prev && prev.verdict === p.verdict && prev.fingerprint === p.fingerprint ? prev.streak + 1 : 1
|
|
340
|
+
memory.set(p.sessionId, {
|
|
341
|
+
verdict: p.verdict,
|
|
342
|
+
confidence: typeof p.confidence === "number" ? p.confidence : null,
|
|
343
|
+
fingerprint: p.fingerprint ?? null,
|
|
344
|
+
streak,
|
|
345
|
+
ts: e.ts ?? null,
|
|
346
|
+
judgedBy: p.judgedBy ?? null,
|
|
347
|
+
})
|
|
348
|
+
}
|
|
349
|
+
return memory
|
|
350
|
+
}
|
|
351
|
+
|
|
352
|
+
/**
|
|
353
|
+
* Cache decision: a session already judged the SAME verdict on the SAME
|
|
354
|
+
* evidence fingerprint for at least `stablePasses` consecutive passes is not
|
|
355
|
+
* re-judged. Any evidence change (fingerprint differs) re-opens it.
|
|
356
|
+
*/
|
|
357
|
+
export function shouldRejudge(memory, sessionId, fingerprint, opts = {}) {
|
|
358
|
+
const stablePasses = opts.stablePasses ?? 2
|
|
359
|
+
const m = memory instanceof Map ? memory.get(sessionId) : memory?.[sessionId]
|
|
360
|
+
if (!m) return { rejudge: true, cached: null }
|
|
361
|
+
if (m.fingerprint !== fingerprint) return { rejudge: true, cached: null }
|
|
362
|
+
if ((m.streak ?? 0) >= stablePasses) return { rejudge: false, cached: m }
|
|
363
|
+
return { rejudge: true, cached: null }
|
|
364
|
+
}
|
|
365
|
+
|
|
366
|
+
/** The `app_state` event to append for one recorded verdict. */
|
|
367
|
+
export function verdictMemoryEvent(input = {}) {
|
|
368
|
+
const payload = {
|
|
369
|
+
kind: "steward-verdict",
|
|
370
|
+
sessionId: input.sessionId,
|
|
371
|
+
verdict: input.verdict,
|
|
372
|
+
confidence: input.confidence ?? null,
|
|
373
|
+
fingerprint: input.fingerprint ?? null,
|
|
374
|
+
...(input.judgedBy ? { judgedBy: input.judgedBy } : {}),
|
|
375
|
+
...(input.note ? { note: input.note } : {}),
|
|
376
|
+
}
|
|
377
|
+
return {
|
|
378
|
+
kind: "note",
|
|
379
|
+
by: "policy",
|
|
380
|
+
stage: "session-steward",
|
|
381
|
+
item: input.sessionId,
|
|
382
|
+
payload,
|
|
383
|
+
}
|
|
384
|
+
}
|
|
385
|
+
|
|
386
|
+
/**
|
|
387
|
+
* An operator disagreement with a recorded verdict — logged as a Jev example
|
|
388
|
+
* (mission item 10: "les désaccords opérateur sont journalisés comme
|
|
389
|
+
* exemples"). Returns `null` when the operator agrees (or there is no prior).
|
|
390
|
+
*/
|
|
391
|
+
export function operatorDisagreement(memory, sessionId, operatorVerdict, opts = {}) {
|
|
392
|
+
const m = memory instanceof Map ? memory.get(sessionId) : memory?.[sessionId]
|
|
393
|
+
if (!m || !operatorVerdict || m.verdict === operatorVerdict) return null
|
|
394
|
+
return {
|
|
395
|
+
sessionId,
|
|
396
|
+
judgeVerdict: m.verdict,
|
|
397
|
+
judgeConfidence: m.confidence,
|
|
398
|
+
operatorVerdict,
|
|
399
|
+
fingerprint: m.fingerprint,
|
|
400
|
+
...(opts.at ? { at: opts.at } : {}),
|
|
401
|
+
}
|
|
402
|
+
}
|
|
403
|
+
|
|
404
|
+
// ── host saturation (mission item 9) ─────────────────────────────────────
|
|
405
|
+
|
|
406
|
+
/** Load / core ratio above which the host counts as saturated. */
|
|
407
|
+
export const SATURATION_LOAD_PER_CORE = 4
|
|
408
|
+
/** Swap used above this percent counts as saturated. */
|
|
409
|
+
export const SATURATION_SWAP_PERCENT = 50
|
|
410
|
+
/** Available RAM below this percent of total counts as saturated. */
|
|
411
|
+
export const SATURATION_FREE_PERCENT = 15
|
|
412
|
+
|
|
413
|
+
/**
|
|
414
|
+
* Is the host saturated, by the same thresholds `host_load`'s own warnings
|
|
415
|
+
* use? Returns the reasons so the report can say why.
|
|
416
|
+
*/
|
|
417
|
+
export function hostSaturation(hostLoad) {
|
|
418
|
+
if (!hostLoad) return { critical: false, reasons: [] }
|
|
419
|
+
const reasons = []
|
|
420
|
+
const perCore = typeof hostLoad.loadPerCore === "number" ? hostLoad.loadPerCore : null
|
|
421
|
+
if (perCore !== null && perCore > SATURATION_LOAD_PER_CORE) {
|
|
422
|
+
reasons.push(`load ${hostLoad.loadAvg?.[0]?.toFixed?.(1) ?? "?"} = ${perCore}x cores (> ${SATURATION_LOAD_PER_CORE}x)`)
|
|
423
|
+
}
|
|
424
|
+
const swapPct = hostLoad.swap?.percent
|
|
425
|
+
if (typeof swapPct === "number" && swapPct > SATURATION_SWAP_PERCENT) reasons.push(`swap ${Math.round(swapPct)}% used`)
|
|
426
|
+
const mem = hostLoad.memory
|
|
427
|
+
if (mem && typeof mem.totalBytes === "number" && mem.totalBytes > 0) {
|
|
428
|
+
const availPct = ((mem.availableBytes ?? mem.freeBytes ?? 0) / mem.totalBytes) * 100
|
|
429
|
+
if (availPct < SATURATION_FREE_PERCENT) reasons.push(`RAM available ${availPct.toFixed(1)}% (< ${SATURATION_FREE_PERCENT}%)`)
|
|
430
|
+
}
|
|
431
|
+
return { critical: reasons.length > 0, reasons }
|
|
432
|
+
}
|
|
433
|
+
|
|
434
|
+
const fmtBytes = (n) => {
|
|
435
|
+
if (typeof n !== "number" || !Number.isFinite(n)) return "?"
|
|
436
|
+
const mb = n / (1024 * 1024)
|
|
437
|
+
return mb >= 1024 ? `${(mb / 1024).toFixed(1)} GB` : `${Math.round(mb)} MB`
|
|
438
|
+
}
|
|
439
|
+
|
|
440
|
+
/** Processes the log wants surfaced under saturation: reparented orphans,
|
|
441
|
+
* then big processes that are NOT owned by a session (acting on sessions
|
|
442
|
+
* alone fixes nothing). Report-only — no process is ever touched here. */
|
|
443
|
+
export function saturationHeader(hostLoad) {
|
|
444
|
+
if (!hostLoad) return []
|
|
445
|
+
const { critical, reasons } = hostSaturation(hostLoad)
|
|
446
|
+
if (!critical) return []
|
|
447
|
+
const lines = [`## ⚠ Host saturated — orphans & non-session processes first (report only)`, ""]
|
|
448
|
+
lines.push(`- ${reasons.join(" · ")}`)
|
|
449
|
+
const proc = (p) => `pid ${p.pid} ${p.command} (${fmtBytes(p.memoryBytes ?? p.rssBytes)}${p.elapsedSec ? `, ${Math.round(p.elapsedSec / 3600)}h` : ""})`
|
|
450
|
+
const orphans = (hostLoad.processes ?? hostLoad.topByMemory ?? []).filter((p) => p?.owner?.kind === "orphan")
|
|
451
|
+
const others = (hostLoad.processes ?? hostLoad.topByMemory ?? []).filter(
|
|
452
|
+
(p) => p?.owner?.kind === "other" || p?.owner?.kind === "system" || p?.owner?.kind === "provisioning",
|
|
453
|
+
)
|
|
454
|
+
if (orphans.length > 0) lines.push(`- orphans: ${orphans.slice(0, 5).map(proc).join("; ")}`)
|
|
455
|
+
if (others.length > 0) lines.push(`- big non-session: ${others.slice(0, 5).map(proc).join("; ")}`)
|
|
456
|
+
const criticalWarnings = (hostLoad.warnings ?? []).filter((w) => w?.severity === "critical")
|
|
457
|
+
if (criticalWarnings.length > 0) lines.push(`- warnings: ${criticalWarnings.map((w) => w.message).join("; ")}`)
|
|
458
|
+
lines.push("")
|
|
459
|
+
return lines
|
|
460
|
+
}
|
|
461
|
+
|
|
462
|
+
// ── explicit 0-candidate report (mission item 8) ─────────────────────────
|
|
463
|
+
|
|
464
|
+
/**
|
|
465
|
+
* Why there is nothing to do, stated explicitly instead of an empty report:
|
|
466
|
+
* how many live sessions were seen, how many were busy, how many were
|
|
467
|
+
* terminal (and whether any still need a relabel), and how many were
|
|
468
|
+
* excluded (self / same cron job / pinned / pty / keepAlive).
|
|
469
|
+
*/
|
|
470
|
+
export function explainZeroCandidates(counts = {}) {
|
|
471
|
+
const n = (v) => (typeof v === "number" ? v : 0)
|
|
472
|
+
return (
|
|
473
|
+
`0 candidates: ${n(counts.live)} live (` +
|
|
474
|
+
`${n(counts.busy)} busy, ${n(counts.idle)} idle), ` +
|
|
475
|
+
`${n(counts.terminal)} terminal (${n(counts.terminalRelabel)} need a relabel), ` +
|
|
476
|
+
`${n(counts.excluded)} excluded (self/cron/pinned/pty/keepAlive)`
|
|
477
|
+
)
|
|
478
|
+
}
|
|
479
|
+
|
|
480
|
+
/** Nothing idle AND nothing terminal-without-outcome ⇒ a full workflow is
|
|
481
|
+
* not worth it; the caller can report and stop. */
|
|
482
|
+
export function shouldFastPath(scan = {}) {
|
|
483
|
+
const busy = (scan.busy ?? []).length
|
|
484
|
+
const idle = (scan.idle ?? []).length
|
|
485
|
+
const terminalRelabel = (scan.terminalRelabel ?? []).length
|
|
486
|
+
const neverRan = (scan.neverRan ?? []).length
|
|
487
|
+
const anyIdleCandidate = idle > 0 || neverRan > 0 || terminalRelabel > 0
|
|
488
|
+
return { fastPath: !anyIdleCandidate && busy > 0, anyIdleCandidate }
|
|
489
|
+
}
|
|
490
|
+
|
|
491
|
+
// ── proposals (mission items 1, 2) ───────────────────────────────────────
|
|
492
|
+
|
|
493
|
+
export const NUDGE_INTERRUPT = "interrupt"
|
|
494
|
+
export const NUDGE_CONTINUE = "continue"
|
|
495
|
+
|
|
496
|
+
/**
|
|
497
|
+
* Merge loop and stall findings into at most ONE nudge proposal per session
|
|
498
|
+
* per pass (interrupt wins over continue: a loop is more specific). A
|
|
499
|
+
* user-origin session is NEVER proposed for a nudge — it is reported as
|
|
500
|
+
* observed-only, so a human stays in the loop. Never proposes a close.
|
|
501
|
+
*/
|
|
502
|
+
export function buildProposals(input = {}) {
|
|
503
|
+
const bySession = new Map()
|
|
504
|
+
const consider = (id, kind, reason, originClass) => {
|
|
505
|
+
if (!id) return
|
|
506
|
+
const existing = bySession.get(id)
|
|
507
|
+
if (existing) {
|
|
508
|
+
if (existing.kind === NUDGE_INTERRUPT) return
|
|
509
|
+
if (kind !== NUDGE_INTERRUPT) return
|
|
510
|
+
}
|
|
511
|
+
bySession.set(id, { sessionId: id, kind, reason, originClass })
|
|
512
|
+
}
|
|
513
|
+
for (const r of input.loopResults ?? []) {
|
|
514
|
+
if (r?.looping) consider(r.sessionId, NUDGE_INTERRUPT, `loop: ${(r.reasons ?? []).join("; ")}`, r.originClass)
|
|
515
|
+
}
|
|
516
|
+
for (const r of input.stallResults ?? []) {
|
|
517
|
+
if (r?.stalled) consider(r.sessionId, NUDGE_CONTINUE, `stall: ${r.reason}`, r.originClass)
|
|
518
|
+
}
|
|
519
|
+
const proposals = []
|
|
520
|
+
const observed = []
|
|
521
|
+
for (const p of bySession.values()) {
|
|
522
|
+
if (p.originClass === "user") observed.push({ ...p, suppressed: "origine utilisateur" })
|
|
523
|
+
else proposals.push(p)
|
|
524
|
+
}
|
|
525
|
+
return { proposals, observed }
|
|
526
|
+
}
|