@agentproto/apps 0.18.0 → 0.20.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.d.ts +2 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.mjs +61 -2
- package/dist/index.mjs.map +1 -1
- package/dist/review-panel/panel.d.ts +1 -1
- package/dist/review-panel/panel.d.ts.map +1 -1
- package/dist/review-panel/panel.generated.d.ts +1 -1
- package/dist/review-panel/panel.generated.d.ts.map +1 -1
- package/dist/review-panel/panel.mjs +1 -1
- package/dist/review-panel/panel.mjs.map +1 -1
- package/dist/review-panel.mjs +1 -1
- package/dist/review-panel.mjs.map +1 -1
- package/dist/store/index.d.ts +61 -0
- package/dist/store/index.d.ts.map +1 -0
- package/dist/store/panel.d.ts +53 -0
- package/dist/store/panel.d.ts.map +1 -0
- package/dist/store/panel.generated.d.ts +13 -0
- package/dist/store/panel.generated.d.ts.map +1 -0
- package/dist/store/panel.mjs +25 -0
- package/dist/store/panel.mjs.map +1 -0
- package/dist/store.mjs +71 -0
- package/dist/store.mjs.map +1 -0
- package/package.json +14 -2
- package/session-steward/.agentproto/APP.md +2 -0
- package/session-steward/.agentproto/workflows/session-steward/WORKFLOW.md +218 -11
- package/session-steward/.agentproto/workflows/session-steward/cron-rules.mjs +526 -0
- package/session-steward/.agentproto/workflows/session-steward/entry.mjs +539 -64
- package/session-steward/.agentproto/workflows/session-steward/origin-policy.mjs +115 -0
- package/session-steward/README.md +19 -2
- package/session-steward/routines/session-steward-hourly/ROUTINE.md +24 -1
- package/session-steward/scripts/sessions-snapshot.sh +85 -0
- package/session-steward/skill/SKILL.md +74 -0
|
@@ -11,16 +11,56 @@
|
|
|
11
11
|
// Safety model: every mutation goes through `session_wrapup_apply`, which
|
|
12
12
|
// RE-CLASSIFIES each id itself immediately before acting and refuses
|
|
13
13
|
// `keep`-class ids outright. On top of that, this workflow:
|
|
14
|
-
// - mutates
|
|
15
|
-
// empty list otherwise)
|
|
14
|
+
// - mutates no SESSION unless `apply` is true (every session-mutating map
|
|
15
|
+
// runs over an empty list otherwise). The one dry-run write is the
|
|
16
|
+
// append-only verdict-memory ledger (`app_state_append`) — never a
|
|
17
|
+
// session, and disableable with `appId: ""`;
|
|
16
18
|
// - only ever feeds `close`/`stuck` ids to the rules pass — a `keepAlive`
|
|
17
19
|
// session can only ever be `judge` class, so rules never close it;
|
|
18
20
|
// - drops the caller's own session from every candidate list;
|
|
19
21
|
// - treats a malformed judge reply as `active` with confidence 0 (never
|
|
20
|
-
// acted on)
|
|
22
|
+
// acted on);
|
|
23
|
+
// - PROPOSES loop/stall nudges in the report only — never sends one, never
|
|
24
|
+
// closes a `looping` session.
|
|
25
|
+
//
|
|
26
|
+
// The mechanical rules themselves live in `cron-rules.mjs` (pure, unit
|
|
27
|
+
// tested): loop detection, stall, never-ran, fast-path done, terminal
|
|
28
|
+
// relabel, self-exclusion, re-check, host saturation, and the verdict-memory
|
|
29
|
+
// fold. This file wires them over `session_list` / `tool_calls_list` /
|
|
30
|
+
// `host_load` / `app_state`.
|
|
31
|
+
|
|
32
|
+
import {
|
|
33
|
+
classifyOrigin,
|
|
34
|
+
decideAction,
|
|
35
|
+
resolveOriginPolicy,
|
|
36
|
+
DEFAULT_CLOSABLE_ORIGINS,
|
|
37
|
+
DEFAULT_USER_ORIGINS,
|
|
38
|
+
} from "./origin-policy.mjs"
|
|
39
|
+
import {
|
|
40
|
+
buildProposals,
|
|
41
|
+
detectLoop,
|
|
42
|
+
detectStall,
|
|
43
|
+
evidenceFingerprint,
|
|
44
|
+
explainZeroCandidates,
|
|
45
|
+
foldVerdictMemory,
|
|
46
|
+
isNeverRan,
|
|
47
|
+
isSelfExcluded,
|
|
48
|
+
saturationHeader,
|
|
49
|
+
shouldRejudge,
|
|
50
|
+
terminalRelabelCandidate,
|
|
51
|
+
verdictMemoryEvent,
|
|
52
|
+
NUDGE_CONTINUE,
|
|
53
|
+
NUDGE_INTERRUPT,
|
|
54
|
+
} from "./cron-rules.mjs"
|
|
21
55
|
|
|
22
56
|
const DEFAULT_IDLE_MINUTES = 30
|
|
23
57
|
const DEFAULT_MIN_CONFIDENCE = 0.8
|
|
58
|
+
/** How many consecutive passes on an unchanged fingerprint before the judge
|
|
59
|
+
* cache stops re-judging a session. */
|
|
60
|
+
const DEFAULT_STABLE_VERDICT_PASSES = 2
|
|
61
|
+
/** The installed app whose `app_state` ledger holds the verdict memory — the
|
|
62
|
+
* `id` in APP.md, which is what the daemon's app registry keys installs by. */
|
|
63
|
+
const DEFAULT_APP_ID = "@agentproto/session-steward"
|
|
24
64
|
// The agent judge's model is the `judge.session` model ROLE, resolved at run
|
|
25
65
|
// time by the `modelRoles` step (the daemon's `model_roles` tool): explicit
|
|
26
66
|
// `judgeModel` input > repo agentproto.json `models` > daemon config `models`
|
|
@@ -37,7 +77,6 @@ const ASK_POLL_MS = 45_000
|
|
|
37
77
|
|
|
38
78
|
const JUDGE_REF = "@agentproto/session-steward-judge"
|
|
39
79
|
const VERDICTS = ["done", "abandoned", "blocked", "needs-input", "active"]
|
|
40
|
-
const APPLY_VERDICTS = new Set(["done", "abandoned", "blocked", "needs-input"])
|
|
41
80
|
|
|
42
81
|
export const ASK_PROMPT =
|
|
43
82
|
"Steward check: is your task complete? Reply exactly `STEWARD: DONE <one line>` " +
|
|
@@ -61,6 +100,7 @@ function explicitOrRole(explicit, modelRoles, role) {
|
|
|
61
100
|
* never a raw `$input.*` that may be absent. */
|
|
62
101
|
export function resolveSettings(input, modelRoles) {
|
|
63
102
|
const i = input ?? {}
|
|
103
|
+
const originPolicy = resolveOriginPolicy({ userOrigins: i.userOrigins, closableOrigins: i.closableOrigins })
|
|
64
104
|
return {
|
|
65
105
|
idleMinutes: Math.floor(num(i.idleMinutes, DEFAULT_IDLE_MINUTES, { min: 1 })),
|
|
66
106
|
apply: i.apply === true,
|
|
@@ -71,9 +111,34 @@ export function resolveSettings(input, modelRoles) {
|
|
|
71
111
|
maxJudged: Math.floor(num(i.maxJudged, DEFAULT_MAX_JUDGED)),
|
|
72
112
|
askSessions: i.askSessions === true,
|
|
73
113
|
callerSessionId: typeof i.callerSessionId === "string" && i.callerSessionId ? i.callerSessionId : null,
|
|
114
|
+
callerOrigin: typeof i.callerOrigin === "string" && i.callerOrigin ? i.callerOrigin : null,
|
|
115
|
+
appId: typeof i.appId === "string" ? i.appId.trim() : DEFAULT_APP_ID,
|
|
116
|
+
stableVerdictPasses: Math.floor(num(i.stableVerdictPasses, DEFAULT_STABLE_VERDICT_PASSES, { min: 1 })),
|
|
117
|
+
userOrigins: originPolicy.userOrigins,
|
|
118
|
+
closableOrigins: originPolicy.closableOrigins,
|
|
74
119
|
}
|
|
75
120
|
}
|
|
76
121
|
|
|
122
|
+
/** The origin policy a settings object carries, as `decideAction` wants it. */
|
|
123
|
+
function policyOf(settings) {
|
|
124
|
+
return { userOrigins: settings?.userOrigins, closableOrigins: settings?.closableOrigins }
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/** `decideAction` over one candidate entry, with the run's policy folded in.
|
|
128
|
+
* Used by both the apply-queue builders and the report, so the action shown
|
|
129
|
+
* and the action executed can never drift. */
|
|
130
|
+
export function decideFor(entry, planClass, verdict, confidence, settings) {
|
|
131
|
+
return decideAction({
|
|
132
|
+
session: entry,
|
|
133
|
+
planClass,
|
|
134
|
+
verdict,
|
|
135
|
+
confidence,
|
|
136
|
+
apply: settings?.apply === true,
|
|
137
|
+
policy: policyOf(settings),
|
|
138
|
+
minConfidence: settings?.minConfidence,
|
|
139
|
+
})
|
|
140
|
+
}
|
|
141
|
+
|
|
77
142
|
// ── plan → candidates ────────────────────────────────────────────────────
|
|
78
143
|
|
|
79
144
|
/** Split `session_wrapup_plan`'s entries into this run's work lists. `keep`
|
|
@@ -97,21 +162,26 @@ export function splitCandidates(planResult, settings) {
|
|
|
97
162
|
}
|
|
98
163
|
}
|
|
99
164
|
|
|
100
|
-
/** Rules pass: `close` → done, `stuck` → abandoned — only when `apply
|
|
165
|
+
/** Rules pass: `close` → done, `stuck` → abandoned — only when `apply`, and
|
|
166
|
+
* bounded by origin. A user-origin candidate is never closed: it is queued
|
|
167
|
+
* as a `needs-input` FLAG instead, with the "origine utilisateur" reason. */
|
|
101
168
|
export function buildRuleApplyQueue(candidates, settings) {
|
|
102
169
|
if (!settings?.apply) return []
|
|
103
|
-
|
|
104
|
-
|
|
105
|
-
|
|
106
|
-
|
|
107
|
-
|
|
108
|
-
|
|
109
|
-
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
}
|
|
114
|
-
|
|
170
|
+
const queue = []
|
|
171
|
+
const push = (entries, planClass, closeVerdict, closeNote) => {
|
|
172
|
+
for (const e of entries ?? []) {
|
|
173
|
+
const d = decideFor(e, planClass, closeVerdict, 1, settings)
|
|
174
|
+
if (d.action === "skip") continue
|
|
175
|
+
queue.push({
|
|
176
|
+
sessionId: e.sessionId,
|
|
177
|
+
verdict: d.action === "close" ? closeVerdict : "needs-input",
|
|
178
|
+
note: d.action === "close" ? closeNote(e) : d.reason,
|
|
179
|
+
})
|
|
180
|
+
}
|
|
181
|
+
}
|
|
182
|
+
push(candidates?.close, "close", "done", e => `steward-rules: ${(e.reasons ?? []).join("; ") || "close class"}`)
|
|
183
|
+
push(candidates?.stuck, "stuck", "abandoned", () => "stuck starting, never ran")
|
|
184
|
+
return queue
|
|
115
185
|
}
|
|
116
186
|
|
|
117
187
|
// ── evidence ─────────────────────────────────────────────────────────────
|
|
@@ -128,9 +198,14 @@ function cut(text, max) {
|
|
|
128
198
|
|
|
129
199
|
/** Plan entry + `session_evidence` → the compact object the judge sees,
|
|
130
200
|
* under {@link EVIDENCE_MAX_CHARS} once serialized (oldest turns dropped
|
|
131
|
-
* first, then the tail signal shortened).
|
|
132
|
-
|
|
201
|
+
* first, then the tail signal shortened). `memory` (a folded verdict map
|
|
202
|
+
* from `app_state`, optional) adds the previous verdict for this session.
|
|
203
|
+
* Every field added by the PR-3 enrichment is copied through only when the
|
|
204
|
+
* `session_evidence` tool supplied it — an old daemon still yields the old
|
|
205
|
+
* shape. */
|
|
206
|
+
export function composeEvidence(entry, raw, memory) {
|
|
133
207
|
const signals = entry?.signals ?? {}
|
|
208
|
+
const previous = memory instanceof Map ? memory.get(entry.sessionId) : memory?.[entry?.sessionId]
|
|
134
209
|
const evidence = {
|
|
135
210
|
sessionId: entry.sessionId,
|
|
136
211
|
label: raw?.label ?? entry.label,
|
|
@@ -142,6 +217,8 @@ export function composeEvidence(entry, raw) {
|
|
|
142
217
|
busy: raw?.busy === true,
|
|
143
218
|
rssMB: mb(entry.rssBytes),
|
|
144
219
|
planReasons: entry.reasons ?? [],
|
|
220
|
+
origin: raw?.origin ?? entry.origin,
|
|
221
|
+
parentSessionId: raw?.parentSessionId ?? entry.parentSessionId,
|
|
145
222
|
signals: {
|
|
146
223
|
lastAssistantTail: cut(signals.lastAssistantTail, 800),
|
|
147
224
|
pendingToolCall: signals.pendingToolCall === true,
|
|
@@ -150,6 +227,22 @@ export function composeEvidence(entry, raw) {
|
|
|
150
227
|
},
|
|
151
228
|
worktree: raw?.worktree ?? null,
|
|
152
229
|
turns: Array.isArray(raw?.turns) ? [...raw.turns] : [],
|
|
230
|
+
...(raw?.liveChildren !== undefined ? { liveChildren: raw.liveChildren } : {}),
|
|
231
|
+
...(raw?.continuedFrom ? { continuedFrom: raw.continuedFrom } : {}),
|
|
232
|
+
...(raw?.continuedTo ? { continuedTo: raw.continuedTo } : {}),
|
|
233
|
+
...(raw?.tokensIn !== undefined ? { tokensIn: raw.tokensIn } : {}),
|
|
234
|
+
...(raw?.tokensOut !== undefined ? { tokensOut: raw.tokensOut } : {}),
|
|
235
|
+
...(raw?.lastTurnErroredAt ? { lastTurnErroredAt: raw.lastTurnErroredAt } : {}),
|
|
236
|
+
...(raw?.lastTurnError ? { lastTurnError: raw.lastTurnError } : {}),
|
|
237
|
+
...(raw?.outcome ? { outcome: raw.outcome } : {}),
|
|
238
|
+
...(raw?.pullRequests ? { pullRequests: raw.pullRequests } : {}),
|
|
239
|
+
...(raw?.toolStats ? { toolStats: raw.toolStats } : {}),
|
|
240
|
+
...(raw?.lastToolCall ? { lastToolCall: raw.lastToolCall } : {}),
|
|
241
|
+
...(raw?.minutesSinceUserMessage !== undefined ? { minutesSinceUserMessage: raw.minutesSinceUserMessage } : {}),
|
|
242
|
+
...(raw?.minutesSinceAgentMessage !== undefined ? { minutesSinceAgentMessage: raw.minutesSinceAgentMessage } : {}),
|
|
243
|
+
...(previous
|
|
244
|
+
? { previousVerdict: { verdict: previous.verdict, confidence: previous.confidence ?? null, streak: previous.streak ?? 1, ts: previous.ts ?? null } }
|
|
245
|
+
: {}),
|
|
153
246
|
}
|
|
154
247
|
while (JSON.stringify(evidence).length > EVIDENCE_MAX_CHARS && evidence.turns.length > 0) evidence.turns.shift()
|
|
155
248
|
if (JSON.stringify(evidence).length > EVIDENCE_MAX_CHARS) evidence.signals.lastAssistantTail = cut(evidence.signals.lastAssistantTail, 200)
|
|
@@ -160,15 +253,22 @@ export function buildJudgePrompt(evidence) {
|
|
|
160
253
|
return (
|
|
161
254
|
"You are the session steward's judge. Decide whether ONE idle AI coding-agent session " +
|
|
162
255
|
"is finished, from the evidence below. Do NOT call any tool — answer from the evidence alone.\n\n" +
|
|
163
|
-
"Verdicts:\n" +
|
|
164
|
-
"- `done`:
|
|
165
|
-
"or
|
|
166
|
-
"
|
|
167
|
-
"
|
|
168
|
-
"
|
|
169
|
-
"- `
|
|
170
|
-
"
|
|
171
|
-
"
|
|
256
|
+
"Verdicts (concrete signals — see the evidence fields named in each):\n" +
|
|
257
|
+
"- `done`: finished with nothing pending — `pullRequests.merged` > 0 or " +
|
|
258
|
+
"`worktree.pr.state`=\"merged\"; or `pullRequests.opened` > 0 with a final report and no " +
|
|
259
|
+
"open question; or the last tool call is a `message_parent` with `kind:\"done\"`; or " +
|
|
260
|
+
"`outcome.verdict`=\"done\"; or the user's last message is an acknowledgement with no " +
|
|
261
|
+
"pending question.\n" +
|
|
262
|
+
"- `abandoned`: superseded or a dead end — `outcome.verdict`=\"abandoned\"/\"failed\", or " +
|
|
263
|
+
"the worktree is gone/merged elsewhere with no open PR and no pending question.\n" +
|
|
264
|
+
"- `blocked`: waiting on something EXTERNAL — an open PR with CI/review pending, " +
|
|
265
|
+
"`liveChildren` > 0, or a `lastTurnError` that clears on its own.\n" +
|
|
266
|
+
"- `needs-input`: waiting on a HUMAN — `awaitingInput` true, or the LAST assistant turn " +
|
|
267
|
+
"ends in a question to the user/operator.\n" +
|
|
268
|
+
"- `active`: mid-work — `busy`, a progress update with no conclusion, recent distinct " +
|
|
269
|
+
"`toolStats`, or an unchanged `previousVerdict` of active.\n" +
|
|
270
|
+
"When the evidence is thin or ambiguous, say `active` with a LOW confidence. Closing a " +
|
|
271
|
+
"session that still had work is worse than leaving an idle one open.\n\n" +
|
|
172
272
|
"Reply with ONLY one JSON object, no prose, no code fence:\n" +
|
|
173
273
|
`{"sessionId": "${evidence.sessionId}", "verdict": "done"|"abandoned"|"blocked"|"needs-input"|"active", ` +
|
|
174
274
|
'"confidence": <number 0..1>, "reason": "<one line>"}\n\n' +
|
|
@@ -184,7 +284,7 @@ function foldEvidence(b) {
|
|
|
184
284
|
if (!raw || raw.sessionId !== b.item?.sessionId) {
|
|
185
285
|
throw new Error(`session_evidence answered for '${raw?.sessionId}', expected '${b.item?.sessionId}'`)
|
|
186
286
|
}
|
|
187
|
-
const evidence = composeEvidence(b.item, raw)
|
|
287
|
+
const evidence = composeEvidence(b.item, raw, b.steps.memory)
|
|
188
288
|
return { entry: b.item, evidence, judgePrompt: buildJudgePrompt(evidence) }
|
|
189
289
|
}
|
|
190
290
|
|
|
@@ -380,18 +480,25 @@ export function mergeDeclared(verdicts, askQueue, askResult) {
|
|
|
380
480
|
// ── judged apply ─────────────────────────────────────────────────────────
|
|
381
481
|
|
|
382
482
|
/** `done`/`abandoned` (close) and `blocked`/`needs-input` (flag) at or above
|
|
383
|
-
* `minConfidence` — only when `apply
|
|
384
|
-
*
|
|
483
|
+
* `minConfidence` — only when `apply`, and bounded by origin: a user-origin
|
|
484
|
+
* candidate is downgraded to a `needs-input` FLAG, never a close. A malformed
|
|
485
|
+
* reply is `active`/0 and can never qualify. */
|
|
385
486
|
export function buildJudgedApplyQueue(finalVerdicts, settings) {
|
|
386
487
|
if (!settings?.apply) return []
|
|
387
|
-
|
|
388
|
-
|
|
389
|
-
|
|
488
|
+
const queue = []
|
|
489
|
+
for (const r of finalVerdicts ?? []) {
|
|
490
|
+
if (r.malformed) continue
|
|
491
|
+
const d = decideFor(r.entry, "judge", r.verdict, r.confidence, settings)
|
|
492
|
+
if (d.action === "skip") continue
|
|
493
|
+
const isFlagVerdict = r.verdict === "blocked" || r.verdict === "needs-input"
|
|
494
|
+
queue.push({
|
|
390
495
|
sessionId: r.entry.sessionId,
|
|
391
|
-
verdict: r.verdict,
|
|
496
|
+
verdict: d.action === "close" ? r.verdict : isFlagVerdict ? r.verdict : "needs-input",
|
|
392
497
|
judgedBy: r.source === "declared" ? `steward-ask:${r.entry.sessionId}` : r.judgedBy ?? r.judgeSessionId ?? "steward-judge",
|
|
393
|
-
note: r.reason
|
|
394
|
-
})
|
|
498
|
+
note: d.action === "close" ? r.reason : `${d.reason}${r.reason ? ` — ${r.reason}` : ""}`,
|
|
499
|
+
})
|
|
500
|
+
}
|
|
501
|
+
return queue
|
|
395
502
|
}
|
|
396
503
|
|
|
397
504
|
// ── report ───────────────────────────────────────────────────────────────
|
|
@@ -418,15 +525,33 @@ function cell(s) {
|
|
|
418
525
|
return String(s ?? "").replace(/\|/g, "\\|").replace(/\n/g, " ")
|
|
419
526
|
}
|
|
420
527
|
|
|
421
|
-
function actionOf(applied
|
|
422
|
-
|
|
423
|
-
|
|
424
|
-
|
|
528
|
+
function actionOf(applied) {
|
|
529
|
+
return applied.ok ? applied.action ?? "applied" : `refused (${applied.error})`
|
|
530
|
+
}
|
|
531
|
+
|
|
532
|
+
/** The `origin` column: the provenance label plus `(user)` when the origin
|
|
533
|
+
* policy bounds this candidate to flag-only. `(none, user)` is a root with
|
|
534
|
+
* no origin and no parent — human-launched, never closed. */
|
|
535
|
+
function originCell(entry, settings) {
|
|
536
|
+
const origin = entry?.origin
|
|
537
|
+
const userBound = classifyOrigin(entry, policyOf(settings)) === "user"
|
|
538
|
+
if (!origin) return userBound ? "(none, user)" : "(none)"
|
|
539
|
+
return userBound ? `${origin} (user)` : origin
|
|
540
|
+
}
|
|
541
|
+
|
|
542
|
+
/** The `action` column: what actually happened (apply) or the retained action
|
|
543
|
+
* the policy decided (dry run / skipped). When an outcome exists, the
|
|
544
|
+
* retained-action label stays visible next to it — that label is what
|
|
545
|
+
* carries the origin bound ("flag (origine utilisateur)"). */
|
|
546
|
+
function actionCell(id, decision, applied) {
|
|
547
|
+
const a = applied.get(id)
|
|
548
|
+
if (!a) return decision.reason
|
|
549
|
+
return `${actionOf(a)} — ${decision.reason}`
|
|
425
550
|
}
|
|
426
551
|
|
|
427
552
|
export function buildReport(b) {
|
|
428
553
|
const s = b.steps.settings ?? resolveSettings(b.input)
|
|
429
|
-
const c = b.steps.candidates ?? { close: [], stuck: [], judge: [], judgeOverflow: [] }
|
|
554
|
+
const c = b.steps.candidatesPlus ?? b.steps.candidates ?? { close: [], stuck: [], judge: [], judgeOverflow: [] }
|
|
430
555
|
const verdicts = b.steps.finalVerdicts ?? []
|
|
431
556
|
const applied = collectApplyResults(b.steps.autoApply, b.steps.judgedApply)
|
|
432
557
|
const lines = []
|
|
@@ -437,27 +562,31 @@ export function buildReport(b) {
|
|
|
437
562
|
`(jev \`${s.jevModel}\`, agent \`${s.judgeModel ?? "agent default"}\`)` +
|
|
438
563
|
(s.askSessions ? " · askSessions on" : ""),
|
|
439
564
|
)
|
|
565
|
+
lines.push(`origins: user=${s.userOrigins.join(", ")} · closable=${s.closableOrigins.join(", ")}`)
|
|
566
|
+
// Host saturation header first, report-only (mission item 9).
|
|
567
|
+
for (const line of saturationHeader(b.steps.hostLoad)) lines.push(line)
|
|
440
568
|
if (!s.apply) lines.push("", "_Dry run: nothing was closed or flagged. Re-run with `apply: true` to act._")
|
|
441
569
|
lines.push("")
|
|
442
|
-
lines.push("| class | session | idle | RAM | verdict | confidence | reason | action |")
|
|
443
|
-
lines.push("
|
|
570
|
+
lines.push("| class | session | origin | idle | RAM | verdict | confidence | reason | action |")
|
|
571
|
+
lines.push("|---|---|---|---|---|---|---|---|---|")
|
|
444
572
|
const row = (cls, e, verdict, conf, reason, action) =>
|
|
445
573
|
lines.push(
|
|
446
|
-
`| ${cls} | ${cell(e.label ?? e.sessionId)} | ${e.idleMinutes ?? "?"} min | ${fmtMB(e.rssBytes)} | ` +
|
|
574
|
+
`| ${cls} | ${cell(e.label ?? e.sessionId)} | ${cell(originCell(e, s))} | ${e.idleMinutes ?? "?"} min | ${fmtMB(e.rssBytes)} | ` +
|
|
447
575
|
`${cell(verdict)} | ${conf === undefined ? "—" : conf.toFixed(2)} | ${cell(reason)} | ${cell(action)} |`,
|
|
448
576
|
)
|
|
449
|
-
for (const e of c.close)
|
|
450
|
-
|
|
577
|
+
for (const e of c.close) {
|
|
578
|
+
const d = decideFor(e, "close", "done", 1, s)
|
|
579
|
+
row("close", e, "done (rules)", undefined, (e.reasons ?? []).join("; "), actionCell(e.sessionId, d, applied))
|
|
580
|
+
}
|
|
581
|
+
for (const e of c.stuck) {
|
|
582
|
+
const d = decideFor(e, "stuck", "abandoned", 1, s)
|
|
583
|
+
row("stuck", e, "abandoned (rules)", undefined, (e.reasons ?? []).join("; ") || "stuck starting, never ran", actionCell(e.sessionId, d, applied))
|
|
584
|
+
}
|
|
451
585
|
for (const r of verdicts) {
|
|
452
|
-
const
|
|
453
|
-
const action = applied.get(r.entry.sessionId)
|
|
454
|
-
? actionOf(applied.get(r.entry.sessionId), s.apply, wouldAct)
|
|
455
|
-
: wouldAct
|
|
456
|
-
? s.apply ? "none" : "none (dry run)"
|
|
457
|
-
: "untouched (below threshold or active)"
|
|
586
|
+
const d = decideFor(r.entry, "judge", r.verdict, r.confidence, s)
|
|
458
587
|
const by = r.source === "declared" ? " (declared)" : r.source === "jev" ? " (jev)" : r.source === "judged" ? " (agent)" : ""
|
|
459
588
|
const reason = r.jevFallback ? `${r.reason} [jev failed: ${r.jevFallback} → agent judge]` : r.reason
|
|
460
|
-
row("judge", r.entry, `${r.verdict}${by}`, r.confidence, reason,
|
|
589
|
+
row("judge", r.entry, `${r.verdict}${by}`, r.confidence, reason, actionCell(r.entry.sessionId, d, applied))
|
|
461
590
|
}
|
|
462
591
|
for (const e of c.judgeOverflow ?? []) row("judge", e, "—", undefined, `not judged this run (maxJudged ${s.maxJudged})`, "none")
|
|
463
592
|
lines.push("")
|
|
@@ -489,9 +618,288 @@ export function buildReport(b) {
|
|
|
489
618
|
}
|
|
490
619
|
lines.push(`- RAM freed (closed sessions): ${fmtMB(freed)}`)
|
|
491
620
|
lines.push(`- RAM still held by idle sessions: ${fmtMB(held)}`)
|
|
621
|
+
|
|
622
|
+
// Explicit "0 candidates" explanation (mission item 8) — say WHY, instead
|
|
623
|
+
// of leaving an empty table to interpret.
|
|
624
|
+
const scan = b.steps.scan
|
|
625
|
+
if (scan && c.close.length === 0 && c.stuck.length === 0 && verdicts.length === 0) {
|
|
626
|
+
lines.push(`- ${explainZeroCandidates(scan.counts)}`)
|
|
627
|
+
}
|
|
628
|
+
|
|
629
|
+
// Nudge proposals (mission items 1-2) — report only, NEVER executed here.
|
|
630
|
+
const prop = b.steps.proposals ?? { proposals: [], observed: [] }
|
|
631
|
+
if ((prop.proposals ?? []).length > 0 || (prop.observed ?? []).length > 0) {
|
|
632
|
+
lines.push("", "## Proposals (report only — no nudge is sent by this workflow)")
|
|
633
|
+
for (const p of prop.proposals ?? []) lines.push(`- ${p.kind} nudge → ${p.sessionId} — ${p.reason}`)
|
|
634
|
+
for (const p of prop.observed ?? []) lines.push(`- observed → ${p.sessionId} — ${p.reason} (no nudge: ${p.suppressed})`)
|
|
635
|
+
}
|
|
636
|
+
|
|
637
|
+
// Terminal sessions with no outcome (mission item 5).
|
|
638
|
+
const relabel = b.steps.relabelQueue ?? []
|
|
639
|
+
if (relabel.length > 0) {
|
|
640
|
+
lines.push("", "## Terminal sessions missing an outcome (relabel candidates)")
|
|
641
|
+
for (const r of relabel) lines.push(`- ${r.sessionId} → ${r.proposedVerdict} — ${r.reason}`)
|
|
642
|
+
}
|
|
643
|
+
|
|
644
|
+
// Verdict memory / cache (mission item 10).
|
|
645
|
+
const memory = b.steps.memory
|
|
646
|
+
const cachedCount = verdicts.filter(r => r.source === "cache").length
|
|
647
|
+
if (b.steps.memoryApp?.note) {
|
|
648
|
+
lines.push(`- verdict memory: ${b.steps.memoryApp.note}`)
|
|
649
|
+
} else if (memory instanceof Map && memory.size > 0) {
|
|
650
|
+
lines.push(`- verdict memory: ${memory.size} session(s) known` + (cachedCount > 0 ? `, ${cachedCount} served from cache` : ""))
|
|
651
|
+
}
|
|
652
|
+
|
|
492
653
|
return lines.join("\n")
|
|
493
654
|
}
|
|
494
655
|
|
|
656
|
+
// ── live scan, loop/stall, memory (mission items 1-10) ───────────────────
|
|
657
|
+
|
|
658
|
+
const TERMINAL_STATUSES = new Set(["killed", "exited", "error", "stopped", "completed", "failed"])
|
|
659
|
+
|
|
660
|
+
function idleMinutesOf(row, nowMs) {
|
|
661
|
+
const ts = row?.lastActivityAt ?? row?.startedAt
|
|
662
|
+
const ms = ts ? Date.parse(ts) : Number.NaN
|
|
663
|
+
return Number.isFinite(ms) ? Math.max(0, (nowMs - ms) / 60_000) : 0
|
|
664
|
+
}
|
|
665
|
+
|
|
666
|
+
/** `session_list` rows under any of its shapes: the un-paged `{sessions}`
|
|
667
|
+
* wrapper (what a workflow tool step gets), the paged `{items}` envelope, or
|
|
668
|
+
* a bare array. */
|
|
669
|
+
export function liveRowsOf(liveSessions) {
|
|
670
|
+
if (Array.isArray(liveSessions?.sessions)) return liveSessions.sessions
|
|
671
|
+
if (Array.isArray(liveSessions?.items)) return liveSessions.items
|
|
672
|
+
if (Array.isArray(liveSessions)) return liveSessions
|
|
673
|
+
return []
|
|
674
|
+
}
|
|
675
|
+
|
|
676
|
+
/**
|
|
677
|
+
* One deterministic pass over the live `session_list` rows: busy sessions
|
|
678
|
+
* (loop/stall scan), idle sessions (zero-candidate accounting), terminal
|
|
679
|
+
* sessions missing an outcome (relabel candidates), never-ran 0/0 sessions,
|
|
680
|
+
* and everything excluded (self / same cron job / archived / pinned / pty /
|
|
681
|
+
* keepAlive). Pure over the rows + settings + an injected `nowMs`.
|
|
682
|
+
*/
|
|
683
|
+
export function scanLive(liveSessions, settings, nowMs) {
|
|
684
|
+
const rows = liveRowsOf(liveSessions)
|
|
685
|
+
const policy = policyOf(settings)
|
|
686
|
+
const self = settings?.callerSessionId ?? null
|
|
687
|
+
const idleThreshold = settings?.idleMinutes ?? DEFAULT_IDLE_MINUTES
|
|
688
|
+
const busy = []
|
|
689
|
+
const idle = []
|
|
690
|
+
const terminal = []
|
|
691
|
+
const terminalRelabel = []
|
|
692
|
+
const neverRan = []
|
|
693
|
+
const excluded = []
|
|
694
|
+
const loopQueue = []
|
|
695
|
+
const stallInputs = []
|
|
696
|
+
let liveCount = 0
|
|
697
|
+
for (const s of rows) {
|
|
698
|
+
const id = s?.id
|
|
699
|
+
if (!id) continue
|
|
700
|
+
if (self && id === self) {
|
|
701
|
+
excluded.push({ sessionId: id, reason: "caller session" })
|
|
702
|
+
continue
|
|
703
|
+
}
|
|
704
|
+
if (isSelfExcluded(s, { callerSessionId: self, callerOrigin: settings?.callerOrigin }).excluded) {
|
|
705
|
+
excluded.push({ sessionId: id, reason: "same cron job as caller" })
|
|
706
|
+
continue
|
|
707
|
+
}
|
|
708
|
+
if (s.archived === true) {
|
|
709
|
+
excluded.push({ sessionId: id, reason: "archived" })
|
|
710
|
+
continue
|
|
711
|
+
}
|
|
712
|
+
if (s.pinned === true) {
|
|
713
|
+
excluded.push({ sessionId: id, reason: "pinned" })
|
|
714
|
+
continue
|
|
715
|
+
}
|
|
716
|
+
if (s.pty === true) {
|
|
717
|
+
excluded.push({ sessionId: id, reason: "pty" })
|
|
718
|
+
continue
|
|
719
|
+
}
|
|
720
|
+
const originClass = classifyOrigin(s, policy)
|
|
721
|
+
const label = s.label ?? s.name
|
|
722
|
+
const idleMinutes = idleMinutesOf(s, nowMs)
|
|
723
|
+
if (TERMINAL_STATUSES.has(String(s.status ?? ""))) {
|
|
724
|
+
terminal.push({ sessionId: id, origin: s.origin, originClass, label })
|
|
725
|
+
const cand = terminalRelabelCandidate(s)
|
|
726
|
+
if (cand.candidate) {
|
|
727
|
+
terminalRelabel.push({ sessionId: id, origin: s.origin, originClass, label, proposedVerdict: cand.proposedVerdict, reason: cand.reason })
|
|
728
|
+
}
|
|
729
|
+
continue
|
|
730
|
+
}
|
|
731
|
+
if (s.status !== "running" && s.status !== "starting") continue
|
|
732
|
+
liveCount++
|
|
733
|
+
if (s.keepAlive === true) {
|
|
734
|
+
excluded.push({ sessionId: id, reason: "keepAlive" })
|
|
735
|
+
continue
|
|
736
|
+
}
|
|
737
|
+
const row = { sessionId: id, origin: s.origin, originClass, label, idleMinutes, lastTurnErroredAt: s.lastTurnErroredAt ?? null }
|
|
738
|
+
if (isNeverRan(s)) neverRan.push(row)
|
|
739
|
+
if (s.busy === true) {
|
|
740
|
+
busy.push(row)
|
|
741
|
+
loopQueue.push({ sessionId: id, originClass, label })
|
|
742
|
+
stallInputs.push({ sessionId: id, originClass, busy: true, idleMinutes, lastTurnErroredAt: s.lastTurnErroredAt ?? null })
|
|
743
|
+
} else if (idleMinutes >= idleThreshold) {
|
|
744
|
+
idle.push(row)
|
|
745
|
+
}
|
|
746
|
+
}
|
|
747
|
+
const counts = {
|
|
748
|
+
live: liveCount,
|
|
749
|
+
busy: busy.length,
|
|
750
|
+
idle: idle.length,
|
|
751
|
+
terminal: terminal.length,
|
|
752
|
+
terminalRelabel: terminalRelabel.length,
|
|
753
|
+
neverRan: neverRan.length,
|
|
754
|
+
excluded: excluded.length,
|
|
755
|
+
}
|
|
756
|
+
return { busy, idle, terminal, terminalRelabel, neverRan, excluded, loopQueue, stallInputs, counts }
|
|
757
|
+
}
|
|
758
|
+
|
|
759
|
+
/** Fold the never-ran 0/0 sessions into the plan as `stuck` (no judge,
|
|
760
|
+
* whatever the idle) and drop them from the judge queue — mission item 3. */
|
|
761
|
+
export function mergeNeverRan(candidates, scan) {
|
|
762
|
+
const never = scan?.neverRan ?? []
|
|
763
|
+
const neverIds = new Set(never.map(n => n.sessionId))
|
|
764
|
+
const existing = new Set((candidates?.stuck ?? []).map(e => e.sessionId))
|
|
765
|
+
const added = never
|
|
766
|
+
.filter(n => !existing.has(n.sessionId))
|
|
767
|
+
.map(n => ({
|
|
768
|
+
sessionId: n.sessionId,
|
|
769
|
+
...(n.label ? { label: n.label } : {}),
|
|
770
|
+
idleMinutes: Math.round(n.idleMinutes ?? 0),
|
|
771
|
+
class: "stuck",
|
|
772
|
+
reasons: ["0 tokens in/out — never ran"],
|
|
773
|
+
signals: {},
|
|
774
|
+
...(n.origin ? { origin: n.origin } : {}),
|
|
775
|
+
}))
|
|
776
|
+
return {
|
|
777
|
+
...candidates,
|
|
778
|
+
stuck: [...(candidates?.stuck ?? []), ...added],
|
|
779
|
+
judge: (candidates?.judge ?? []).filter(e => !neverIds.has(e.sessionId)),
|
|
780
|
+
judgeOverflow: (candidates?.judgeOverflow ?? []).filter(e => !neverIds.has(e.sessionId)),
|
|
781
|
+
}
|
|
782
|
+
}
|
|
783
|
+
|
|
784
|
+
/** `tool_calls_list` map item → the loop verdict + stats for one session. */
|
|
785
|
+
export function analyzeLoopItem(b) {
|
|
786
|
+
const item = b.item ?? {}
|
|
787
|
+
const raw = b.steps.loopCallsOne
|
|
788
|
+
const records = Array.isArray(raw?.records) ? raw.records : Array.isArray(raw) ? raw : []
|
|
789
|
+
const r = detectLoop(records, { nowMs: Date.now() })
|
|
790
|
+
return { sessionId: item.sessionId, label: item.label, originClass: item.originClass, ...r }
|
|
791
|
+
}
|
|
792
|
+
|
|
793
|
+
/** Stall verdicts for every busy live session. */
|
|
794
|
+
export function analyzeStalls(scan, nowMs) {
|
|
795
|
+
return (scan?.stallInputs ?? []).map(s => ({
|
|
796
|
+
sessionId: s.sessionId,
|
|
797
|
+
originClass: s.originClass,
|
|
798
|
+
...detectStall({ busy: s.busy, idleMinutes: s.idleMinutes, lastTurnErroredAt: s.lastTurnErroredAt, nowMs }),
|
|
799
|
+
}))
|
|
800
|
+
}
|
|
801
|
+
|
|
802
|
+
/** Ids out of an `app_list` result (bare array, `{apps}` or `{items}`). */
|
|
803
|
+
function installedAppIdsOf(appList) {
|
|
804
|
+
const rows = Array.isArray(appList) ? appList : Array.isArray(appList?.apps) ? appList.apps : Array.isArray(appList?.items) ? appList.items : []
|
|
805
|
+
return rows.map(r => (typeof r === "string" ? r : r?.appId ?? r?.id)).filter(id => typeof id === "string" && id)
|
|
806
|
+
}
|
|
807
|
+
|
|
808
|
+
/** Which installed app holds the verdict memory. An exact id match wins; a
|
|
809
|
+
* bare name (`session-steward`) also matches a scoped install
|
|
810
|
+
* (`@agentproto/session-steward`). A missing app, or `appId: ""`, is "no
|
|
811
|
+
* memory" with a note for the report — never a failed step. */
|
|
812
|
+
export function resolveMemoryApp(settings, appList) {
|
|
813
|
+
const wanted = settings?.appId
|
|
814
|
+
if (!wanted) return { appId: null, note: "off (appId empty)" }
|
|
815
|
+
const installed = installedAppIdsOf(appList)
|
|
816
|
+
const hit = installed.find(id => id === wanted) ?? installed.find(id => id.endsWith(`/${wanted}`))
|
|
817
|
+
if (hit) return { appId: hit, note: null }
|
|
818
|
+
return { appId: null, note: `off — no installed app "${wanted}" (install it, or pass its installed id as appId); no verdict was read or recorded` }
|
|
819
|
+
}
|
|
820
|
+
|
|
821
|
+
/** Fold the `app_state` read into the per-session verdict memory map. */
|
|
822
|
+
export function foldMemory(b) {
|
|
823
|
+
const events = settled(b.steps.memoryRead).ok.flatMap(r => (Array.isArray(r.value?.events) ? r.value.events : []))
|
|
824
|
+
return foldVerdictMemory(events)
|
|
825
|
+
}
|
|
826
|
+
|
|
827
|
+
/** Judge candidates minus those already judged the same verdict on the same
|
|
828
|
+
* evidence fingerprint for `stableVerdictPasses` passes (the cache). */
|
|
829
|
+
export function buildJudgeQueueFiltered(evidenceResult, memory, settings) {
|
|
830
|
+
const rows = settled(evidenceResult).ok.map(r => r.value)
|
|
831
|
+
const queue = []
|
|
832
|
+
const cached = []
|
|
833
|
+
for (const q of rows) {
|
|
834
|
+
const fingerprint = evidenceFingerprint(q.evidence)
|
|
835
|
+
const decision = shouldRejudge(memory, q.entry.sessionId, fingerprint, { stablePasses: settings?.stableVerdictPasses })
|
|
836
|
+
if (!decision.rejudge && decision.cached) cached.push({ ...q, cached: decision.cached, fingerprint })
|
|
837
|
+
else queue.push(q)
|
|
838
|
+
}
|
|
839
|
+
return { queue, cached }
|
|
840
|
+
}
|
|
841
|
+
|
|
842
|
+
/** Cached rows as verdict rows, so they appear in the report and (when they
|
|
843
|
+
* carry a confident close verdict) can still be applied without re-judging. */
|
|
844
|
+
export function buildCachedVerdicts(cachedQueue) {
|
|
845
|
+
return (cachedQueue ?? []).map(q => ({
|
|
846
|
+
entry: q.entry,
|
|
847
|
+
evidence: q.evidence,
|
|
848
|
+
verdict: q.cached.verdict,
|
|
849
|
+
confidence: typeof q.cached.confidence === "number" ? q.cached.confidence : 0,
|
|
850
|
+
reason: `cached verdict (stable ${q.cached.streak ?? "?"} passes, evidence unchanged)`,
|
|
851
|
+
source: "cache",
|
|
852
|
+
cached: true,
|
|
853
|
+
judgedBy: q.cached.judgedBy ?? "steward-cache",
|
|
854
|
+
}))
|
|
855
|
+
}
|
|
856
|
+
|
|
857
|
+
/** `collectVerdicts` + the cached rows (cache rows are never re-judged). */
|
|
858
|
+
export function buildVerdicts(b) {
|
|
859
|
+
const jq = b.steps.judgeQueue ?? {}
|
|
860
|
+
return [
|
|
861
|
+
...collectVerdicts(b.steps.evidence, jq.queue, b.steps.jevJudge, b.steps.jevQueue, b.steps.agentJudgeQueue, b.steps.judge),
|
|
862
|
+
...buildCachedVerdicts(jq.cached),
|
|
863
|
+
]
|
|
864
|
+
}
|
|
865
|
+
|
|
866
|
+
/** The report's nudge PROPOSALS (loop → interrupt, stall → continue) plus
|
|
867
|
+
* the user-origin findings reported as observed-only. Never a close. */
|
|
868
|
+
export function buildProposalsStep(scan, loopResults, settings, nowMs) {
|
|
869
|
+
const stalls = analyzeStalls(scan, nowMs)
|
|
870
|
+
const { proposals, observed } = buildProposals({ loopResults, stallResults: stalls })
|
|
871
|
+
return { proposals, observed, stalls }
|
|
872
|
+
}
|
|
873
|
+
|
|
874
|
+
/** Terminal sessions missing an outcome, as relabel candidates. */
|
|
875
|
+
export function buildRelabelQueue(scan) {
|
|
876
|
+
return (scan?.terminalRelabel ?? []).map(t => ({ ...t }))
|
|
877
|
+
}
|
|
878
|
+
|
|
879
|
+
/** The `app_state` events to append for this pass's verdicts. The memory is
|
|
880
|
+
* written on every pass (it is a ledger, never a session action) so streaks
|
|
881
|
+
* accumulate and the cache can engage. */
|
|
882
|
+
export function buildMemoryWriteQueue(finalVerdicts, settings) {
|
|
883
|
+
if (!settings?.appId) return []
|
|
884
|
+
const out = []
|
|
885
|
+
for (const r of finalVerdicts ?? []) {
|
|
886
|
+
if (!r?.entry?.sessionId || r.malformed) continue
|
|
887
|
+
const fingerprint = r.evidence ? evidenceFingerprint(r.evidence) : null
|
|
888
|
+
out.push({
|
|
889
|
+
appId: settings.appId,
|
|
890
|
+
event: verdictMemoryEvent({
|
|
891
|
+
sessionId: r.entry.sessionId,
|
|
892
|
+
verdict: r.verdict,
|
|
893
|
+
confidence: r.confidence,
|
|
894
|
+
fingerprint,
|
|
895
|
+
judgedBy: r.judgedBy ?? r.source ?? null,
|
|
896
|
+
note: r.reason,
|
|
897
|
+
}),
|
|
898
|
+
})
|
|
899
|
+
}
|
|
900
|
+
return out
|
|
901
|
+
}
|
|
902
|
+
|
|
495
903
|
// ── the workflow ─────────────────────────────────────────────────────────
|
|
496
904
|
|
|
497
905
|
export default {
|
|
@@ -502,6 +910,8 @@ export default {
|
|
|
502
910
|
"`close`/`stuck` sessions, judge the ambiguous `judge` ones with a cheap " +
|
|
503
911
|
"one-shot model over compact evidence, optionally ask a session directly, " +
|
|
504
912
|
"then close or flag the confident verdicts with a recorded outcome — and report. " +
|
|
913
|
+
"Origin-bounded: a human-launched session (`chat-starter`, `vscode`, or a " +
|
|
914
|
+
"root with no origin and no parent) is only ever flagged, never closed. " +
|
|
505
915
|
"Dry run unless `apply` is true.",
|
|
506
916
|
version: "0.1.0",
|
|
507
917
|
inputs: {
|
|
@@ -514,6 +924,11 @@ export default {
|
|
|
514
924
|
maxJudged: { type: "number", description: `Most \`judge\` sessions judged per run, most RAM first. Default ${DEFAULT_MAX_JUDGED}.`, default: DEFAULT_MAX_JUDGED },
|
|
515
925
|
askSessions: { type: "boolean", description: "Ask low-confidence idle sessions directly whether they're done. Default false — it spends a turn in someone else's conversation.", default: false },
|
|
516
926
|
callerSessionId: { type: "string", description: "The calling session's id — never a candidate. The CLI passes AGENTPROTO_SESSION_ID." },
|
|
927
|
+
callerOrigin: { type: "string", description: "The calling session's origin (`cron:<jobId>`) — an older run of the SAME cron job is never judged as user work." },
|
|
928
|
+
appId: { type: "string", description: `Installed app whose \`app_state\` ledger holds the verdict memory. Default ${DEFAULT_APP_ID}.` },
|
|
929
|
+
stableVerdictPasses: { type: "number", description: `Consecutive passes on an unchanged evidence fingerprint before the judge cache stops re-judging. Default ${DEFAULT_STABLE_VERDICT_PASSES}.`, default: DEFAULT_STABLE_VERDICT_PASSES },
|
|
930
|
+
userOrigins: { type: "array", description: `Origins that are ALWAYS flag-only, never closed (a human is in the loop). Trailing \`*\` is a prefix wildcard. Default ${JSON.stringify(DEFAULT_USER_ORIGINS)}.`, items: { type: "string" }, default: DEFAULT_USER_ORIGINS },
|
|
931
|
+
closableOrigins: { type: "array", description: `Origins that may be closed under the current rules (cron jobs, gates). Trailing \`*\` is a prefix wildcard. Executors (a session with a parentSessionId) are closable regardless. Default ${JSON.stringify(DEFAULT_CLOSABLE_ORIGINS)}.`, items: { type: "string" }, default: DEFAULT_CLOSABLE_ORIGINS },
|
|
517
932
|
},
|
|
518
933
|
outputs: {},
|
|
519
934
|
steps: [
|
|
@@ -528,10 +943,17 @@ export default {
|
|
|
528
943
|
id: "plan",
|
|
529
944
|
kind: "tool",
|
|
530
945
|
tool: "session_wrapup_plan",
|
|
531
|
-
inputs: { idleMinutes: "$steps.settings.idleMinutes" },
|
|
946
|
+
inputs: { idleMinutes: "$steps.settings.idleMinutes", wait: true },
|
|
532
947
|
},
|
|
533
948
|
{ id: "candidates", kind: "transform", compute: b => splitCandidates(b.steps.plan, b.steps.settings) },
|
|
534
|
-
|
|
949
|
+
// Host saturation header (report only — mission item 9) and the live
|
|
950
|
+
// session scan behind loop/stall/never-ran/terminal rules (items 1-5).
|
|
951
|
+
{ id: "hostLoad", kind: "tool", tool: "host_load", inputs: {} },
|
|
952
|
+
{ id: "liveSessions", kind: "tool", tool: "session_list", inputs: { full: true } },
|
|
953
|
+
{ id: "scan", kind: "transform", compute: b => scanLive(b.steps.liveSessions, b.steps.settings, Date.now()) },
|
|
954
|
+
// Never-ran 0/0 sessions are `stuck` immediately, never judged (item 3).
|
|
955
|
+
{ id: "candidatesPlus", kind: "transform", compute: b => mergeNeverRan(b.steps.candidates, b.steps.scan) },
|
|
956
|
+
{ id: "ruleApplyQueue", kind: "transform", compute: b => buildRuleApplyQueue(b.steps.candidatesPlus, b.steps.settings) },
|
|
535
957
|
{
|
|
536
958
|
// Empty unless `apply` — a dry run dispatches no apply call at all.
|
|
537
959
|
id: "autoApply",
|
|
@@ -548,10 +970,48 @@ export default {
|
|
|
548
970
|
},
|
|
549
971
|
],
|
|
550
972
|
},
|
|
973
|
+
// Verdict memory (item 10): read the app_state ledger best-effort. The
|
|
974
|
+
// map is empty when no app id is set, so a caller can turn memory off.
|
|
975
|
+
{ id: "installedApps", kind: "tool", tool: "app_list", inputs: {} },
|
|
976
|
+
{ id: "memoryApp", kind: "transform", compute: b => resolveMemoryApp(b.steps.settings, b.steps.installedApps) },
|
|
977
|
+
{ id: "memoryQueue", kind: "transform", compute: b => (b.steps.memoryApp?.appId ? [{ appId: b.steps.memoryApp.appId }] : []) },
|
|
978
|
+
{
|
|
979
|
+
id: "memoryRead",
|
|
980
|
+
kind: "map",
|
|
981
|
+
over: "$steps.memoryQueue",
|
|
982
|
+
parallelism: 1,
|
|
983
|
+
onError: "collect",
|
|
984
|
+
steps: [
|
|
985
|
+
{
|
|
986
|
+
id: "memoryReadOne",
|
|
987
|
+
kind: "tool",
|
|
988
|
+
tool: "app_state_list",
|
|
989
|
+
inputs: { appId: "$item.appId", stage: "session-steward", kinds: ["note"], limit: 500 },
|
|
990
|
+
},
|
|
991
|
+
],
|
|
992
|
+
},
|
|
993
|
+
{ id: "memory", kind: "transform", compute: foldMemory },
|
|
994
|
+
// Loop sanity over the busy sessions (item 1) — one tool_calls_list each.
|
|
995
|
+
{
|
|
996
|
+
id: "loopScan",
|
|
997
|
+
kind: "map",
|
|
998
|
+
over: "$steps.scan.loopQueue",
|
|
999
|
+
parallelism: 4,
|
|
1000
|
+
onError: "collect",
|
|
1001
|
+
steps: [
|
|
1002
|
+
{ id: "loopCallsOne", kind: "tool", tool: "tool_calls_list", inputs: { sessionId: "$item.sessionId", lastN: 60 } },
|
|
1003
|
+
{ id: "loopFold", kind: "transform", compute: analyzeLoopItem },
|
|
1004
|
+
],
|
|
1005
|
+
},
|
|
1006
|
+
{ id: "loopResults", kind: "transform", compute: b => settled(b.steps.loopScan).ok.map(r => r.value) },
|
|
1007
|
+
// Nudge PROPOSALS (never executed here): loop → interrupt, stall →
|
|
1008
|
+
// continue, at most one per session per pass, user origins observed only.
|
|
1009
|
+
{ id: "proposals", kind: "transform", compute: b => buildProposalsStep(b.steps.scan, b.steps.loopResults, b.steps.settings, Date.now()) },
|
|
1010
|
+
{ id: "relabelQueue", kind: "transform", compute: b => buildRelabelQueue(b.steps.scan) },
|
|
551
1011
|
{
|
|
552
1012
|
id: "evidence",
|
|
553
1013
|
kind: "map",
|
|
554
|
-
over: "$steps.
|
|
1014
|
+
over: "$steps.candidatesPlus.judge",
|
|
555
1015
|
parallelism: 4,
|
|
556
1016
|
onError: "collect",
|
|
557
1017
|
steps: [
|
|
@@ -561,11 +1021,12 @@ export default {
|
|
|
561
1021
|
{ id: "evidenceFold", kind: "transform", compute: foldEvidence },
|
|
562
1022
|
],
|
|
563
1023
|
},
|
|
564
|
-
|
|
1024
|
+
// Judge queue minus sessions cached by stable verdict+fingerprint (item 10).
|
|
1025
|
+
{ id: "judgeQueue", kind: "transform", compute: b => buildJudgeQueueFiltered(b.steps.evidence, b.steps.memory, b.steps.settings) },
|
|
565
1026
|
{
|
|
566
1027
|
id: "jevQueue",
|
|
567
1028
|
kind: "transform",
|
|
568
|
-
compute: b => (b.steps.settings?.judge === "agent" ? [] : b.steps.judgeQueue ?? []),
|
|
1029
|
+
compute: b => (b.steps.settings?.judge === "agent" ? [] : b.steps.judgeQueue?.queue ?? []),
|
|
569
1030
|
},
|
|
570
1031
|
{
|
|
571
1032
|
// Jev backend: one calibrated `choice` call per candidate. Never an
|
|
@@ -588,7 +1049,7 @@ export default {
|
|
|
588
1049
|
{
|
|
589
1050
|
id: "agentJudgeQueue",
|
|
590
1051
|
kind: "transform",
|
|
591
|
-
compute: b => buildAgentJudgeQueue(b.steps.judgeQueue, b.steps.jevQueue, b.steps.jevJudge, b.steps.settings),
|
|
1052
|
+
compute: b => buildAgentJudgeQueue(b.steps.judgeQueue?.queue, b.steps.jevQueue, b.steps.jevJudge, b.steps.settings),
|
|
592
1053
|
},
|
|
593
1054
|
{
|
|
594
1055
|
// One-shot judge per candidate. The engine releases (kills + archives)
|
|
@@ -617,9 +1078,7 @@ export default {
|
|
|
617
1078
|
},
|
|
618
1079
|
],
|
|
619
1080
|
},
|
|
620
|
-
{ id: "verdicts", kind: "transform", compute:
|
|
621
|
-
collectVerdicts(b.steps.evidence, b.steps.judgeQueue, b.steps.jevJudge, b.steps.jevQueue, b.steps.agentJudgeQueue, b.steps.judge),
|
|
622
|
-
},
|
|
1081
|
+
{ id: "verdicts", kind: "transform", compute: buildVerdicts },
|
|
623
1082
|
{ id: "askQueue", kind: "transform", compute: b => buildAskQueue(b.steps.verdicts, b.steps.settings) },
|
|
624
1083
|
{
|
|
625
1084
|
// Empty unless `askSessions`. ONE prompt per session (queue:false — a
|
|
@@ -689,14 +1148,30 @@ export default {
|
|
|
689
1148
|
},
|
|
690
1149
|
],
|
|
691
1150
|
},
|
|
1151
|
+
// Verdict memory write-back (item 10) — a ledger append, never a session
|
|
1152
|
+
// action; best-effort (an uninstalled app just yields no memory).
|
|
1153
|
+
{ id: "memoryWriteQueue", kind: "transform", compute: b => buildMemoryWriteQueue(b.steps.finalVerdicts, { ...b.steps.settings, appId: b.steps.memoryApp?.appId ?? "" }) },
|
|
1154
|
+
{
|
|
1155
|
+
id: "memoryWrite",
|
|
1156
|
+
kind: "map",
|
|
1157
|
+
over: "$steps.memoryWriteQueue",
|
|
1158
|
+
parallelism: 1,
|
|
1159
|
+
onError: "collect",
|
|
1160
|
+
steps: [
|
|
1161
|
+
{ id: "memoryWriteOne", kind: "tool", tool: "app_state_append", inputs: { appId: "$item.appId", event: "$item.event" } },
|
|
1162
|
+
],
|
|
1163
|
+
},
|
|
692
1164
|
{ id: "report", kind: "transform", compute: b => buildReport(b) },
|
|
693
1165
|
],
|
|
694
1166
|
result: {
|
|
695
1167
|
report: "$steps.report",
|
|
696
1168
|
apply: "$steps.settings.apply",
|
|
697
|
-
candidates: "$steps.
|
|
1169
|
+
candidates: "$steps.candidatesPlus",
|
|
698
1170
|
verdicts: "$steps.finalVerdicts",
|
|
699
1171
|
autoApply: "$steps.autoApply",
|
|
700
1172
|
judgedApply: "$steps.judgedApply",
|
|
1173
|
+
proposals: "$steps.proposals",
|
|
1174
|
+
relabel: "$steps.relabelQueue",
|
|
1175
|
+
scan: "$steps.scan",
|
|
701
1176
|
},
|
|
702
1177
|
}
|