@agentproto/apps 0.15.0 → 0.17.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,705 @@
1
+ // Runtime source of truth for WORKFLOW.md's step graph (the same pattern as
2
+ // repo-maintenance's `maintain/entry.mjs`) — the frontmatter mirrors this by
3
+ // id+kind for governance (`reconcileEntry` only checks the top-level id/kind
4
+ // sequence, never nested step bodies).
5
+ //
6
+ // Entry-based because every decision between the tool calls is a real
7
+ // function: splitting the plan, capping/ordering judge candidates, parsing the
8
+ // judge's JSON strictly, thresholding on `minConfidence`, and the report. The
9
+ // declarative manifest has no expression language for `compute`.
10
+ //
11
+ // Safety model: every mutation goes through `session_wrapup_apply`, which
12
+ // RE-CLASSIFIES each id itself immediately before acting and refuses
13
+ // `keep`-class ids outright. On top of that, this workflow:
14
+ // - mutates nothing unless `apply` is true (every mutating map runs over an
15
+ // empty list otherwise);
16
+ // - only ever feeds `close`/`stuck` ids to the rules pass — a `keepAlive`
17
+ // session can only ever be `judge` class, so rules never close it;
18
+ // - drops the caller's own session from every candidate list;
19
+ // - treats a malformed judge reply as `active` with confidence 0 (never
20
+ // acted on).
21
+
22
+ import { tmpdir } from "node:os"
23
+
24
+ const DEFAULT_IDLE_MINUTES = 30
25
+ const DEFAULT_MIN_CONFIDENCE = 0.8
26
+ // The agent judge's model is the `judge.session` model ROLE, resolved at run
27
+ // time by the `modelRoles` step (the daemon's `model_roles` tool): explicit
28
+ // `judgeModel` input > repo agentproto.json `models` > daemon config `models`
29
+ // > built-in default (packages/runtime/src/model-roles.ts). No model id here.
30
+ const ROLE_JUDGE_SESSION = "judge.session"
31
+ const DEFAULT_MAX_JUDGED = 15
32
+ const DEFAULT_JEV_MODEL = "jev-latest"
33
+ const JUDGE_BACKENDS = ["auto", "jev", "agent"]
34
+ /** Evidence JSON cap per session, in characters. */
35
+ const EVIDENCE_MAX_CHARS = 5_000
36
+ /** `ask`: 4 × 45 s long-polls ≈ 3 min bounded wait for the session's reply. */
37
+ const ASK_WAIT_POLLS = 4
38
+ const ASK_POLL_MS = 45_000
39
+
40
+ const JUDGE_REF = "@agentproto/session-steward-judge"
41
+ const VERDICTS = ["done", "abandoned", "blocked", "needs-input", "active"]
42
+ const APPLY_VERDICTS = new Set(["done", "abandoned", "blocked", "needs-input"])
43
+
44
+ export const ASK_PROMPT =
45
+ "Steward check: is your task complete? Reply exactly `STEWARD: DONE <one line>` " +
46
+ "or `STEWARD: NOT-DONE <one line>`."
47
+
48
+ // ── settings ─────────────────────────────────────────────────────────────
49
+
50
+ function num(v, fallback, { min = 0, max = Number.POSITIVE_INFINITY } = {}) {
51
+ return typeof v === "number" && Number.isFinite(v) && v >= min && v <= max ? v : fallback
52
+ }
53
+
54
+ /** An explicit input model, else the `modelRoles` step's resolution of `role`;
55
+ * undefined leaves the judge agent's own AGENT.md `model` in charge. */
56
+ function explicitOrRole(explicit, modelRoles, role) {
57
+ if (typeof explicit === "string" && explicit.trim()) return explicit.trim()
58
+ const resolved = modelRoles?.models?.[role]
59
+ return typeof resolved === "string" && resolved ? resolved : undefined
60
+ }
61
+
62
+ /** Every input with its default applied — steps read `$steps.settings.*`,
63
+ * never a raw `$input.*` that may be absent. */
64
+ export function resolveSettings(input, modelRoles) {
65
+ const i = input ?? {}
66
+ return {
67
+ idleMinutes: Math.floor(num(i.idleMinutes, DEFAULT_IDLE_MINUTES, { min: 1 })),
68
+ apply: i.apply === true,
69
+ minConfidence: num(i.minConfidence, DEFAULT_MIN_CONFIDENCE, { max: 1 }),
70
+ judgeModel: explicitOrRole(i.judgeModel, modelRoles, ROLE_JUDGE_SESSION),
71
+ judge: JUDGE_BACKENDS.includes(i.judge) ? i.judge : "auto",
72
+ jevModel: typeof i.jevModel === "string" && i.jevModel.trim() ? i.jevModel.trim() : DEFAULT_JEV_MODEL,
73
+ maxJudged: Math.floor(num(i.maxJudged, DEFAULT_MAX_JUDGED)),
74
+ askSessions: i.askSessions === true,
75
+ callerSessionId: typeof i.callerSessionId === "string" && i.callerSessionId ? i.callerSessionId : null,
76
+ }
77
+ }
78
+
79
+ // ── plan → candidates ────────────────────────────────────────────────────
80
+
81
+ /** Split `session_wrapup_plan`'s entries into this run's work lists. `keep`
82
+ * entries (only present if the plan was asked for them) and the caller's own
83
+ * session are dropped here, whatever the plan said. `judge` is ordered most
84
+ * RAM first and capped at `maxJudged`; the rest are counted as overflow. */
85
+ export function splitCandidates(planResult, settings) {
86
+ const entries = Array.isArray(planResult?.entries) ? planResult.entries : []
87
+ const self = settings?.callerSessionId ?? null
88
+ const usable = entries.filter(e => e && typeof e.sessionId === "string" && e.class !== "keep" && e.sessionId !== self)
89
+ const byClass = cls => usable.filter(e => e.class === cls)
90
+ const judgeAll = byClass("judge").sort((a, b) => (b.rssBytes ?? 0) - (a.rssBytes ?? 0))
91
+ const cap = settings?.maxJudged ?? DEFAULT_MAX_JUDGED
92
+ return {
93
+ close: byClass("close"),
94
+ stuck: byClass("stuck"),
95
+ judge: judgeAll.slice(0, cap),
96
+ judgeOverflow: judgeAll.slice(cap),
97
+ keepSkipped: entries.filter(e => e?.class === "keep").length,
98
+ selfExcluded: self !== null && entries.some(e => e?.sessionId === self),
99
+ }
100
+ }
101
+
102
+ /** Rules pass: `close` → done, `stuck` → abandoned — only when `apply`. */
103
+ export function buildRuleApplyQueue(candidates, settings) {
104
+ if (!settings?.apply) return []
105
+ return [
106
+ ...(candidates?.close ?? []).map(e => ({
107
+ sessionId: e.sessionId,
108
+ verdict: "done",
109
+ note: `steward-rules: ${(e.reasons ?? []).join("; ") || "close class"}`,
110
+ })),
111
+ ...(candidates?.stuck ?? []).map(e => ({
112
+ sessionId: e.sessionId,
113
+ verdict: "abandoned",
114
+ note: "stuck starting, never ran",
115
+ })),
116
+ ]
117
+ }
118
+
119
+ // ── evidence ─────────────────────────────────────────────────────────────
120
+
121
+ function mb(bytes) {
122
+ return typeof bytes === "number" ? Math.round(bytes / (1024 * 1024)) : undefined
123
+ }
124
+
125
+ function cut(text, max) {
126
+ if (typeof text !== "string") return undefined
127
+ const t = text.trim()
128
+ return t.length <= max ? t : `…${t.slice(t.length - (max - 1))}`
129
+ }
130
+
131
+ /** Plan entry + `session_evidence` → the compact object the judge sees,
132
+ * under {@link EVIDENCE_MAX_CHARS} once serialized (oldest turns dropped
133
+ * first, then the tail signal shortened). */
134
+ export function composeEvidence(entry, raw) {
135
+ const signals = entry?.signals ?? {}
136
+ const evidence = {
137
+ sessionId: entry.sessionId,
138
+ label: raw?.label ?? entry.label,
139
+ cwd: raw?.cwd,
140
+ adapter: raw?.adapter,
141
+ idleMinutes: entry.idleMinutes,
142
+ keepAlive: raw?.keepAlive === true,
143
+ awaitingInput: raw?.awaitingInput === true,
144
+ busy: raw?.busy === true,
145
+ rssMB: mb(entry.rssBytes),
146
+ planReasons: entry.reasons ?? [],
147
+ signals: {
148
+ lastAssistantTail: cut(signals.lastAssistantTail, 800),
149
+ pendingToolCall: signals.pendingToolCall === true,
150
+ parentEnded: signals.parentEnded === true,
151
+ worktreeMerged: signals.worktreeMerged === true,
152
+ },
153
+ worktree: raw?.worktree ?? null,
154
+ turns: Array.isArray(raw?.turns) ? [...raw.turns] : [],
155
+ }
156
+ while (JSON.stringify(evidence).length > EVIDENCE_MAX_CHARS && evidence.turns.length > 0) evidence.turns.shift()
157
+ if (JSON.stringify(evidence).length > EVIDENCE_MAX_CHARS) evidence.signals.lastAssistantTail = cut(evidence.signals.lastAssistantTail, 200)
158
+ return evidence
159
+ }
160
+
161
+ export function buildJudgePrompt(evidence) {
162
+ return (
163
+ "You are the session steward's judge. Decide whether ONE idle AI coding-agent session " +
164
+ "is finished, from the evidence below. Do NOT call any tool — answer from the evidence alone.\n\n" +
165
+ "Verdicts:\n" +
166
+ "- `done`: the task visibly finished — a PR was opened or merged, a final report was given, " +
167
+ "or the user said thanks/ok with nothing pending.\n" +
168
+ "- `abandoned`: superseded or a dead end, with nothing worth keeping.\n" +
169
+ "- `blocked`: waiting on something external (CI, another session, a dependency).\n" +
170
+ "- `needs-input`: waiting on a human answer or decision.\n" +
171
+ "- `active`: mid-work — keep it.\n" +
172
+ "When unsure, say `active` with a LOW confidence. Closing a session that still had work " +
173
+ "is worse than leaving an idle one open.\n\n" +
174
+ "Reply with ONLY one JSON object, no prose, no code fence:\n" +
175
+ `{"sessionId": "${evidence.sessionId}", "verdict": "done"|"abandoned"|"blocked"|"needs-input"|"active", ` +
176
+ '"confidence": <number 0..1>, "reason": "<one line>"}\n\n' +
177
+ `Evidence:\n${JSON.stringify(evidence)}`
178
+ )
179
+ }
180
+
181
+ /** Fold one evidence map item: the `session_evidence` answer read off the
182
+ * step slot this item's tool step just wrote. The id check guards against
183
+ * ever pairing one session's evidence with another's plan entry. */
184
+ function foldEvidence(b) {
185
+ const raw = b.steps.evidenceOne
186
+ if (!raw || raw.sessionId !== b.item?.sessionId) {
187
+ throw new Error(`session_evidence answered for '${raw?.sessionId}', expected '${b.item?.sessionId}'`)
188
+ }
189
+ const evidence = composeEvidence(b.item, raw)
190
+ return { entry: b.item, evidence, judgePrompt: buildJudgePrompt(evidence) }
191
+ }
192
+
193
+ /** Items of a tolerant (`onError: collect`) map, split by outcome. */
194
+ function settled(mapResult) {
195
+ if (Array.isArray(mapResult)) return { ok: mapResult.map((value, index) => ({ index, value })), failed: [] }
196
+ const results = Array.isArray(mapResult?.results) ? mapResult.results : []
197
+ return {
198
+ ok: results.filter(r => r?.status === "fulfilled").map(r => ({ index: r.index, value: r.value })),
199
+ failed: results.filter(r => r && r.status !== "fulfilled"),
200
+ }
201
+ }
202
+
203
+ // ── judge ────────────────────────────────────────────────────────────────
204
+
205
+ /** Strict parse of the judge's reply. ANYTHING off — not a lone JSON object,
206
+ * wrong/missing sessionId, unknown verdict, confidence outside 0..1, no
207
+ * reason — is `active` with confidence 0, which is never acted on. */
208
+ export function parseJudgeVerdict(text, sessionId) {
209
+ const malformed = why => ({ verdict: "active", confidence: 0, reason: `malformed judge reply: ${why}`, malformed: true })
210
+ if (typeof text !== "string" || !text.trim()) return malformed("empty")
211
+ let body = text.trim()
212
+ const fence = body.match(/^```(?:json)?\s*([\s\S]*?)\s*```$/)
213
+ if (fence) body = fence[1].trim()
214
+ if (!body.startsWith("{") || !body.endsWith("}")) return malformed("not a lone JSON object")
215
+ let v
216
+ try {
217
+ v = JSON.parse(body)
218
+ } catch {
219
+ return malformed("invalid JSON")
220
+ }
221
+ if (!v || typeof v !== "object" || Array.isArray(v)) return malformed("not an object")
222
+ if (v.sessionId !== sessionId) return malformed("sessionId mismatch")
223
+ if (!VERDICTS.includes(v.verdict)) return malformed("unknown verdict")
224
+ if (typeof v.confidence !== "number" || !Number.isFinite(v.confidence) || v.confidence < 0 || v.confidence > 1) {
225
+ return malformed("confidence not a number in 0..1")
226
+ }
227
+ if (typeof v.reason !== "string" || !v.reason.trim()) return malformed("no reason")
228
+ return { verdict: v.verdict, confidence: v.confidence, reason: v.reason.trim().split("\n")[0].slice(0, 300) }
229
+ }
230
+
231
+ /** Jev answers (`session_judge_jev`) by session id — only the items whose
232
+ * map step ran AND answered for the session it was asked about. */
233
+ function jevAnswers(jevResult, jevQueue) {
234
+ const out = new Map()
235
+ const { ok, failed } = settled(jevResult)
236
+ for (const { index, value } of ok) {
237
+ const q = (jevQueue ?? [])[index]
238
+ if (q && value?.sessionId === q.entry.sessionId) out.set(q.entry.sessionId, value)
239
+ }
240
+ for (const f of failed) {
241
+ const q = (jevQueue ?? [])[f.index]
242
+ if (q) out.set(q.entry.sessionId, { ok: false, sessionId: q.entry.sessionId, error: f.error ?? f.status })
243
+ }
244
+ return out
245
+ }
246
+
247
+ /** Candidates the agent judge takes: all of them with `judge: "agent"`,
248
+ * else those Jev didn't answer. `jevFallback` records why — except `auto`
249
+ * with no key, which is simply the agent backend, not a failure. */
250
+ export function buildAgentJudgeQueue(judgeQueue, jevQueue, jevResult, settings) {
251
+ if (settings?.judge === "agent") return [...(judgeQueue ?? [])]
252
+ const answers = jevAnswers(jevResult, jevQueue)
253
+ const out = []
254
+ for (const q of judgeQueue ?? []) {
255
+ const a = answers.get(q.entry.sessionId)
256
+ if (a?.ok === true) continue
257
+ const quietNoKey = settings?.judge === "auto" && a?.noKey === true
258
+ out.push({ ...q, ...(quietNoKey ? {} : { jevFallback: a?.error ?? "no Jev answer" }) })
259
+ }
260
+ return out
261
+ }
262
+
263
+ function formatProbabilities(p) {
264
+ return Object.entries(p ?? {})
265
+ .sort((a, b) => b[1] - a[1])
266
+ .map(([k, v]) => `${k}=${Number(v).toFixed(2)}`)
267
+ .join(" ")
268
+ }
269
+
270
+ /** One row per judged candidate: Jev's answer, else the agent judge's
271
+ * parsed verdict, else `active`/0 when its turn failed outright.
272
+ * Candidates whose evidence failed are rows too (never judged, never acted
273
+ * on). */
274
+ export function collectVerdicts(evidenceResult, judgeQueue, jevResult, jevQueue, agentJudgeQueue, judgeResult) {
275
+ const rows = []
276
+ const ev = settled(evidenceResult)
277
+ for (const f of ev.failed) {
278
+ const entry = f.item ?? {}
279
+ rows.push({ entry, evidence: null, verdict: "active", confidence: 0, reason: `evidence failed: ${f.error ?? f.status}`, source: "none" })
280
+ }
281
+ const jev = jevAnswers(jevResult, jevQueue)
282
+ const agent = new Map()
283
+ const judged = settled(judgeResult)
284
+ const failedByIndex = new Map(judged.failed.map(r => [r.index, r]))
285
+ ;(agentJudgeQueue ?? []).forEach((q, index) => {
286
+ const v = judged.ok.find(r => r.index === index)?.value
287
+ if (v && v.sessionId === q.entry.sessionId) {
288
+ agent.set(q.entry.sessionId, { ...v, source: "judged" })
289
+ } else {
290
+ const f = failedByIndex.get(index)
291
+ agent.set(q.entry.sessionId, {
292
+ verdict: "active",
293
+ confidence: 0,
294
+ reason: f ? `judge failed: ${f.error ?? f.status}` : "judge produced no verdict",
295
+ source: "none",
296
+ })
297
+ }
298
+ if (q.jevFallback) agent.get(q.entry.sessionId).jevFallback = q.jevFallback
299
+ })
300
+ for (const q of judgeQueue ?? []) {
301
+ const id = q.entry.sessionId
302
+ const j = jev.get(id)
303
+ if (j?.ok === true) {
304
+ rows.push({
305
+ entry: q.entry,
306
+ evidence: q.evidence,
307
+ verdict: j.verdict,
308
+ confidence: j.confidence,
309
+ reason: `jev ${j.verdict} — p: ${formatProbabilities(j.probabilities)}`,
310
+ probabilities: j.probabilities,
311
+ source: "jev",
312
+ judgedBy: `jev:${j.model}`,
313
+ })
314
+ continue
315
+ }
316
+ const a = agent.get(id) ?? { verdict: "active", confidence: 0, reason: "not judged", source: "none" }
317
+ rows.push({ entry: q.entry, evidence: q.evidence, ...a })
318
+ }
319
+ return rows
320
+ }
321
+
322
+ // ── ask (opt-in) ─────────────────────────────────────────────────────────
323
+
324
+ /** Sessions to ask directly: judged below `minConfidence`, and idle, not
325
+ * keepAlive, not awaitingInput — only when `askSessions`. */
326
+ export function buildAskQueue(verdicts, settings) {
327
+ if (!settings?.askSessions) return []
328
+ return (verdicts ?? [])
329
+ .filter(
330
+ r =>
331
+ r.evidence &&
332
+ r.confidence < settings.minConfidence &&
333
+ r.evidence.keepAlive !== true &&
334
+ r.evidence.awaitingInput !== true &&
335
+ r.evidence.busy !== true,
336
+ )
337
+ .map(r => ({ sessionId: r.entry.sessionId }))
338
+ }
339
+
340
+ /** `STEWARD: DONE <line>` / `STEWARD: NOT-DONE <line>` in the session's
341
+ * newest assistant turn. Anything else ⇒ `null` (no declaration). */
342
+ export function parseStewardReply(text) {
343
+ if (typeof text !== "string") return null
344
+ const lines = text.split("\n").map(l => l.trim().replace(/^`+|`+$/g, ""))
345
+ for (let i = lines.length - 1; i >= 0; i--) {
346
+ const m = lines[i].match(/^STEWARD:\s*(DONE|NOT-DONE)\b\s*(.*)$/)
347
+ if (m) return { declared: m[1] === "DONE" ? "done" : "not-done", line: m[2].trim().slice(0, 300) }
348
+ }
349
+ return null
350
+ }
351
+
352
+ function lastAssistantText(evidence) {
353
+ const turns = Array.isArray(evidence?.turns) ? evidence.turns : []
354
+ for (let i = turns.length - 1; i >= 0; i--) if (turns[i]?.role === "assistant") return turns[i].text
355
+ return undefined
356
+ }
357
+
358
+ /** Merge the `ask` answers into the verdict rows: a declared DONE becomes a
359
+ * `done` verdict at confidence 1 (`source: "declared"`); a NOT-DONE becomes
360
+ * `active` at confidence 1 (keep). No/unparseable reply ⇒ row unchanged. */
361
+ export function mergeDeclared(verdicts, askQueue, askResult) {
362
+ const answers = new Map()
363
+ const ok = settled(askResult).ok
364
+ for (const { index, value } of ok) {
365
+ const q = (askQueue ?? [])[index]
366
+ if (q && value?.sessionId === q.sessionId && value.declared) answers.set(q.sessionId, value)
367
+ }
368
+ return (verdicts ?? []).map(r => {
369
+ const a = answers.get(r.entry.sessionId)
370
+ if (!a) return r
371
+ return {
372
+ ...r,
373
+ verdict: a.declared === "done" ? "done" : "active",
374
+ confidence: 1,
375
+ reason: `self-declared ${a.declared.toUpperCase()}: ${a.line || "(no detail)"}`,
376
+ source: "declared",
377
+ judgeVerdict: { verdict: r.verdict, confidence: r.confidence, reason: r.reason },
378
+ }
379
+ })
380
+ }
381
+
382
+ // ── judged apply ─────────────────────────────────────────────────────────
383
+
384
+ /** `done`/`abandoned` (close) and `blocked`/`needs-input` (flag) at or above
385
+ * `minConfidence` — only when `apply`. A malformed reply is `active`/0 and
386
+ * can never qualify. */
387
+ export function buildJudgedApplyQueue(finalVerdicts, settings) {
388
+ if (!settings?.apply) return []
389
+ return (finalVerdicts ?? [])
390
+ .filter(r => !r.malformed && APPLY_VERDICTS.has(r.verdict) && r.confidence >= settings.minConfidence)
391
+ .map(r => ({
392
+ sessionId: r.entry.sessionId,
393
+ verdict: r.verdict,
394
+ judgedBy: r.source === "declared" ? `steward-ask:${r.entry.sessionId}` : r.judgedBy ?? r.judgeSessionId ?? "steward-judge",
395
+ note: r.reason,
396
+ }))
397
+ }
398
+
399
+ // ── report ───────────────────────────────────────────────────────────────
400
+
401
+ /** Per-session `session_wrapup_apply` results across both apply passes. */
402
+ export function collectApplyResults(...mapResults) {
403
+ const out = new Map()
404
+ for (const m of mapResults) {
405
+ for (const { value } of settled(m).ok) {
406
+ for (const r of Array.isArray(value?.results) ? value.results : []) out.set(r.sessionId, r)
407
+ }
408
+ for (const f of settled(m).failed) {
409
+ if (f.item?.sessionId) out.set(f.item.sessionId, { sessionId: f.item.sessionId, ok: false, error: f.error ?? f.status })
410
+ }
411
+ }
412
+ return out
413
+ }
414
+
415
+ function fmtMB(bytes) {
416
+ return typeof bytes === "number" ? `${mb(bytes)} MB` : "?"
417
+ }
418
+
419
+ function cell(s) {
420
+ return String(s ?? "").replace(/\|/g, "\\|").replace(/\n/g, " ")
421
+ }
422
+
423
+ function actionOf(applied, apply, wouldAct) {
424
+ if (applied) return applied.ok ? applied.action ?? "applied" : `refused (${applied.error})`
425
+ if (!apply) return wouldAct ? "none (dry run)" : "none"
426
+ return "none"
427
+ }
428
+
429
+ export function buildReport(b) {
430
+ const s = b.steps.settings ?? resolveSettings(b.input)
431
+ const c = b.steps.candidates ?? { close: [], stuck: [], judge: [], judgeOverflow: [] }
432
+ const verdicts = b.steps.finalVerdicts ?? []
433
+ const applied = collectApplyResults(b.steps.autoApply, b.steps.judgedApply)
434
+ const lines = []
435
+ lines.push(`# Session steward — ${s.apply ? "apply" : "dry run"}`)
436
+ lines.push("")
437
+ lines.push(
438
+ `idle ≥ ${s.idleMinutes} min · minConfidence ${s.minConfidence} · judge \`${s.judge}\` ` +
439
+ `(jev \`${s.jevModel}\`, agent \`${s.judgeModel ?? "agent default"}\`)` +
440
+ (s.askSessions ? " · askSessions on" : ""),
441
+ )
442
+ if (!s.apply) lines.push("", "_Dry run: nothing was closed or flagged. Re-run with `apply: true` to act._")
443
+ lines.push("")
444
+ lines.push("| class | session | idle | RAM | verdict | confidence | reason | action |")
445
+ lines.push("|---|---|---|---|---|---|---|---|")
446
+ const row = (cls, e, verdict, conf, reason, action) =>
447
+ lines.push(
448
+ `| ${cls} | ${cell(e.label ?? e.sessionId)} | ${e.idleMinutes ?? "?"} min | ${fmtMB(e.rssBytes)} | ` +
449
+ `${cell(verdict)} | ${conf === undefined ? "—" : conf.toFixed(2)} | ${cell(reason)} | ${cell(action)} |`,
450
+ )
451
+ for (const e of c.close) row("close", e, "done (rules)", undefined, (e.reasons ?? []).join("; "), actionOf(applied.get(e.sessionId), s.apply, true))
452
+ for (const e of c.stuck) row("stuck", e, "abandoned (rules)", undefined, "stuck starting, never ran", actionOf(applied.get(e.sessionId), s.apply, true))
453
+ for (const r of verdicts) {
454
+ const wouldAct = !r.malformed && APPLY_VERDICTS.has(r.verdict) && r.confidence >= s.minConfidence
455
+ const action = applied.get(r.entry.sessionId)
456
+ ? actionOf(applied.get(r.entry.sessionId), s.apply, wouldAct)
457
+ : wouldAct
458
+ ? s.apply ? "none" : "none (dry run)"
459
+ : "untouched (below threshold or active)"
460
+ const by = r.source === "declared" ? " (declared)" : r.source === "jev" ? " (jev)" : r.source === "judged" ? " (agent)" : ""
461
+ const reason = r.jevFallback ? `${r.reason} [jev failed: ${r.jevFallback} → agent judge]` : r.reason
462
+ row("judge", r.entry, `${r.verdict}${by}`, r.confidence, reason, action)
463
+ }
464
+ for (const e of c.judgeOverflow ?? []) row("judge", e, "—", undefined, `not judged this run (maxJudged ${s.maxJudged})`, "none")
465
+ lines.push("")
466
+
467
+ const all = [...c.close, ...c.stuck, ...verdicts.map(r => r.entry), ...(c.judgeOverflow ?? [])]
468
+ let freed = 0
469
+ let held = 0
470
+ for (const e of all) {
471
+ const a = applied.get(e.sessionId)
472
+ if (a?.ok && a.action === "closed") freed += e.rssBytes ?? 0
473
+ else held += e.rssBytes ?? 0
474
+ }
475
+ const counts = {}
476
+ for (const r of verdicts) counts[r.verdict] = (counts[r.verdict] ?? 0) + 1
477
+ lines.push(
478
+ `- candidates: close=${c.close.length} stuck=${c.stuck.length} judge=${verdicts.length}` +
479
+ (c.judgeOverflow?.length ? ` (+${c.judgeOverflow.length} not judged)` : ""),
480
+ )
481
+ if (verdicts.length > 0) lines.push(`- verdicts: ${Object.entries(counts).map(([k, n]) => `${k}=${n}`).join(" ")}`)
482
+ const bySource = { jev: 0, agent: 0 }
483
+ let fallbacks = 0
484
+ for (const r of verdicts) {
485
+ if (r.source === "jev" || r.judgeVerdict?.reason?.startsWith("jev ")) bySource.jev++
486
+ else if (r.source === "judged" || r.judgeSessionId) bySource.agent++
487
+ if (r.jevFallback) fallbacks++
488
+ }
489
+ if (verdicts.length > 0) {
490
+ lines.push(`- judged by: jev=${bySource.jev} agent=${bySource.agent}` + (fallbacks > 0 ? ` (${fallbacks} Jev failure(s) fell back to the agent judge)` : ""))
491
+ }
492
+ lines.push(`- RAM freed (closed sessions): ${fmtMB(freed)}`)
493
+ lines.push(`- RAM still held by idle sessions: ${fmtMB(held)}`)
494
+ return lines.join("\n")
495
+ }
496
+
497
+ // ── the workflow ─────────────────────────────────────────────────────────
498
+
499
+ export default {
500
+ name: "Session Steward",
501
+ id: "session-steward",
502
+ description:
503
+ "Plan idle-session wrap-up (session_wrapup_plan), close the rule-certain " +
504
+ "`close`/`stuck` sessions, judge the ambiguous `judge` ones with a cheap " +
505
+ "one-shot model over compact evidence, optionally ask a session directly, " +
506
+ "then close or flag the confident verdicts with a recorded outcome — and report. " +
507
+ "Dry run unless `apply` is true.",
508
+ version: "0.1.0",
509
+ inputs: {
510
+ idleMinutes: { type: "number", description: `Idle threshold in minutes. Default ${DEFAULT_IDLE_MINUTES}.`, default: DEFAULT_IDLE_MINUTES },
511
+ apply: { type: "boolean", description: "Close/flag sessions. Default false = dry run (plan + verdicts, no mutation).", default: false },
512
+ minConfidence: { type: "number", description: `Judge confidence needed to act. Default ${DEFAULT_MIN_CONFIDENCE}.`, default: DEFAULT_MIN_CONFIDENCE },
513
+ judge: { type: "string", description: "Judge backend: `auto` (Jev when JEV_API_KEY resolves, else the agent judge), `jev`, or `agent`. A Jev failure always falls back to the agent judge. Default auto.", default: "auto" },
514
+ jevModel: { type: "string", description: `Jev model. Default ${DEFAULT_JEV_MODEL}.`, default: DEFAULT_JEV_MODEL },
515
+ judgeModel: { type: "string", description: `Model for the agent judge. Default: the \`${ROLE_JUDGE_SESSION}\` model role (repo agentproto.json \`models\` > daemon config \`models\` > built-in).` },
516
+ maxJudged: { type: "number", description: `Most \`judge\` sessions judged per run, most RAM first. Default ${DEFAULT_MAX_JUDGED}.`, default: DEFAULT_MAX_JUDGED },
517
+ askSessions: { type: "boolean", description: "Ask low-confidence idle sessions directly whether they're done. Default false — it spends a turn in someone else's conversation.", default: false },
518
+ callerSessionId: { type: "string", description: "The calling session's id — never a candidate. The CLI passes AGENTPROTO_SESSION_ID." },
519
+ },
520
+ outputs: {},
521
+ steps: [
522
+ {
523
+ id: "modelRoles",
524
+ kind: "tool",
525
+ tool: "model_roles",
526
+ inputs: { roles: [ROLE_JUDGE_SESSION], inputs: { [ROLE_JUDGE_SESSION]: "$input.judgeModel" } },
527
+ },
528
+ { id: "settings", kind: "transform", compute: b => resolveSettings(b.input, b.steps.modelRoles) },
529
+ {
530
+ id: "plan",
531
+ kind: "tool",
532
+ tool: "session_wrapup_plan",
533
+ inputs: { idleMinutes: "$steps.settings.idleMinutes" },
534
+ },
535
+ { id: "candidates", kind: "transform", compute: b => splitCandidates(b.steps.plan, b.steps.settings) },
536
+ { id: "ruleApplyQueue", kind: "transform", compute: b => buildRuleApplyQueue(b.steps.candidates, b.steps.settings) },
537
+ {
538
+ // Empty unless `apply` — a dry run dispatches no apply call at all.
539
+ id: "autoApply",
540
+ kind: "map",
541
+ over: "$steps.ruleApplyQueue",
542
+ parallelism: 1,
543
+ onError: "collect",
544
+ steps: [
545
+ {
546
+ id: "autoApplyOne",
547
+ kind: "tool",
548
+ tool: "session_wrapup_apply",
549
+ inputs: { sessionIds: ["$item.sessionId"], verdict: "$item.verdict", note: "$item.note" },
550
+ },
551
+ ],
552
+ },
553
+ {
554
+ id: "evidence",
555
+ kind: "map",
556
+ over: "$steps.candidates.judge",
557
+ parallelism: 4,
558
+ onError: "collect",
559
+ steps: [
560
+ { id: "evidenceOne", kind: "tool", tool: "session_evidence", inputs: { sessionId: "$item.sessionId" } },
561
+ // Reads `$steps.evidenceOne` in the transform right after this item's
562
+ // own tool step wrote it (no await between) and checks the id.
563
+ { id: "evidenceFold", kind: "transform", compute: foldEvidence },
564
+ ],
565
+ },
566
+ { id: "judgeQueue", kind: "transform", compute: b => settled(b.steps.evidence).ok.map(r => r.value) },
567
+ {
568
+ id: "jevQueue",
569
+ kind: "transform",
570
+ compute: b => (b.steps.settings?.judge === "agent" ? [] : b.steps.judgeQueue ?? []),
571
+ },
572
+ {
573
+ // Jev backend: one calibrated `choice` call per candidate. Never an
574
+ // error result — a missing key or failure is `ok:false`, and that
575
+ // candidate goes to the agent judge instead.
576
+ id: "jevJudge",
577
+ kind: "map",
578
+ over: "$steps.jevQueue",
579
+ parallelism: 4,
580
+ onError: "collect",
581
+ steps: [
582
+ {
583
+ id: "jevOne",
584
+ kind: "tool",
585
+ tool: "session_judge_jev",
586
+ inputs: { sessionId: "$item.entry.sessionId", evidence: "$item.evidence", model: "$steps.settings.jevModel" },
587
+ },
588
+ ],
589
+ },
590
+ {
591
+ id: "agentJudgeQueue",
592
+ kind: "transform",
593
+ compute: b => buildAgentJudgeQueue(b.steps.judgeQueue, b.steps.jevQueue, b.steps.jevJudge, b.steps.settings),
594
+ },
595
+ {
596
+ // One-shot judge per candidate. The engine releases (kills + archives)
597
+ // each judge session as soon as its map item settles.
598
+ id: "judge",
599
+ kind: "map",
600
+ over: "$steps.agentJudgeQueue",
601
+ parallelism: 3,
602
+ onError: "collect",
603
+ steps: [
604
+ {
605
+ id: "judgeOne",
606
+ kind: "agent",
607
+ agent: { ref: JUDGE_REF },
608
+ cwd: tmpdir(),
609
+ prompt: "$item.judgePrompt",
610
+ model: b => b.steps.settings?.judgeModel,
611
+ },
612
+ {
613
+ id: "judgeParse",
614
+ kind: "transform",
615
+ compute: b => ({
616
+ sessionId: b.item?.entry?.sessionId,
617
+ judgeSessionId: b.steps.judgeOne?.sessionId,
618
+ ...parseJudgeVerdict(b.steps.judgeOne?.text, b.item?.entry?.sessionId),
619
+ }),
620
+ },
621
+ ],
622
+ },
623
+ { id: "verdicts", kind: "transform", compute: b =>
624
+ collectVerdicts(b.steps.evidence, b.steps.judgeQueue, b.steps.jevJudge, b.steps.jevQueue, b.steps.agentJudgeQueue, b.steps.judge),
625
+ },
626
+ { id: "askQueue", kind: "transform", compute: b => buildAskQueue(b.steps.verdicts, b.steps.settings) },
627
+ {
628
+ // Empty unless `askSessions`. ONE prompt per session (queue:false — a
629
+ // session that turned busy meanwhile refuses it), then a bounded wait,
630
+ // then its newest assistant turn is parsed for the STEWARD line.
631
+ id: "ask",
632
+ kind: "map",
633
+ over: "$steps.askQueue",
634
+ parallelism: 2,
635
+ onError: "collect",
636
+ steps: [
637
+ {
638
+ id: "askPrompt",
639
+ kind: "tool",
640
+ tool: "agent_prompt",
641
+ inputs: { sessionId: "$item.sessionId", prompt: ASK_PROMPT, queue: false, interrupt: false },
642
+ },
643
+ { id: "askArm", kind: "transform", compute: b => ((b.item.askWaiting = true), true) },
644
+ {
645
+ id: "askWait",
646
+ kind: "loop",
647
+ while: "$item.askWaiting",
648
+ max_iterations: ASK_WAIT_POLLS,
649
+ steps: [
650
+ {
651
+ id: "askMonitor",
652
+ kind: "tool",
653
+ tool: "session_monitor",
654
+ inputs: { sessionId: "$item.sessionId", event: "turn-end", timeoutMs: ASK_POLL_MS },
655
+ },
656
+ {
657
+ id: "askMonitorFold",
658
+ kind: "transform",
659
+ compute: b => ((b.item.askWaiting = b.steps.askMonitor?.timedOut === true), b.item.askWaiting),
660
+ },
661
+ ],
662
+ },
663
+ { id: "askRead", kind: "tool", tool: "session_evidence", inputs: { sessionId: "$item.sessionId" } },
664
+ {
665
+ id: "askParse",
666
+ kind: "transform",
667
+ compute: b => {
668
+ const raw = b.steps.askRead
669
+ if (raw?.sessionId !== b.item.sessionId) return { sessionId: b.item.sessionId, declared: null }
670
+ const parsed = parseStewardReply(lastAssistantText(raw))
671
+ return { sessionId: b.item.sessionId, declared: parsed?.declared ?? null, line: parsed?.line ?? "" }
672
+ },
673
+ },
674
+ ],
675
+ },
676
+ { id: "finalVerdicts", kind: "transform", compute: b => mergeDeclared(b.steps.verdicts, b.steps.askQueue, b.steps.ask) },
677
+ { id: "judgedApplyQueue", kind: "transform", compute: b => buildJudgedApplyQueue(b.steps.finalVerdicts, b.steps.settings) },
678
+ {
679
+ // Empty unless `apply`. blocked/needs-input only FLAG (the tool never
680
+ // closes on those); done/abandoned close resumably with the outcome.
681
+ id: "judgedApply",
682
+ kind: "map",
683
+ over: "$steps.judgedApplyQueue",
684
+ parallelism: 1,
685
+ onError: "collect",
686
+ steps: [
687
+ {
688
+ id: "judgedApplyOne",
689
+ kind: "tool",
690
+ tool: "session_wrapup_apply",
691
+ inputs: { sessionIds: ["$item.sessionId"], verdict: "$item.verdict", judgedBy: "$item.judgedBy", note: "$item.note" },
692
+ },
693
+ ],
694
+ },
695
+ { id: "report", kind: "transform", compute: b => buildReport(b) },
696
+ ],
697
+ result: {
698
+ report: "$steps.report",
699
+ apply: "$steps.settings.apply",
700
+ candidates: "$steps.candidates",
701
+ verdicts: "$steps.finalVerdicts",
702
+ autoApply: "$steps.autoApply",
703
+ judgedApply: "$steps.judgedApply",
704
+ },
705
+ }