@agentproto/apps 0.17.2 → 0.19.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -11,16 +11,55 @@
11
11
  // Safety model: every mutation goes through `session_wrapup_apply`, which
12
12
  // RE-CLASSIFIES each id itself immediately before acting and refuses
13
13
  // `keep`-class ids outright. On top of that, this workflow:
14
- // - mutates nothing unless `apply` is true (every mutating map runs over an
15
- // empty list otherwise);
14
+ // - mutates no SESSION unless `apply` is true (every session-mutating map
15
+ // runs over an empty list otherwise). The one dry-run write is the
16
+ // append-only verdict-memory ledger (`app_state_append`) — never a
17
+ // session, and disableable with `appId: ""`;
16
18
  // - only ever feeds `close`/`stuck` ids to the rules pass — a `keepAlive`
17
19
  // session can only ever be `judge` class, so rules never close it;
18
20
  // - drops the caller's own session from every candidate list;
19
21
  // - treats a malformed judge reply as `active` with confidence 0 (never
20
- // acted on).
22
+ // acted on);
23
+ // - PROPOSES loop/stall nudges in the report only — never sends one, never
24
+ // closes a `looping` session.
25
+ //
26
+ // The mechanical rules themselves live in `cron-rules.mjs` (pure, unit
27
+ // tested): loop detection, stall, never-ran, fast-path done, terminal
28
+ // relabel, self-exclusion, re-check, host saturation, and the verdict-memory
29
+ // fold. This file wires them over `session_list` / `tool_calls_list` /
30
+ // `host_load` / `app_state`.
31
+
32
+ import {
33
+ classifyOrigin,
34
+ decideAction,
35
+ resolveOriginPolicy,
36
+ DEFAULT_CLOSABLE_ORIGINS,
37
+ DEFAULT_USER_ORIGINS,
38
+ } from "./origin-policy.mjs"
39
+ import {
40
+ buildProposals,
41
+ detectLoop,
42
+ detectStall,
43
+ evidenceFingerprint,
44
+ explainZeroCandidates,
45
+ foldVerdictMemory,
46
+ isNeverRan,
47
+ isSelfExcluded,
48
+ saturationHeader,
49
+ shouldRejudge,
50
+ terminalRelabelCandidate,
51
+ verdictMemoryEvent,
52
+ NUDGE_CONTINUE,
53
+ NUDGE_INTERRUPT,
54
+ } from "./cron-rules.mjs"
21
55
 
22
56
  const DEFAULT_IDLE_MINUTES = 30
23
57
  const DEFAULT_MIN_CONFIDENCE = 0.8
58
+ /** How many consecutive passes on an unchanged fingerprint before the judge
59
+ * cache stops re-judging a session. */
60
+ const DEFAULT_STABLE_VERDICT_PASSES = 2
61
+ /** The installed app whose `app_state` ledger holds the verdict memory. */
62
+ const DEFAULT_APP_ID = "session-steward"
24
63
  // The agent judge's model is the `judge.session` model ROLE, resolved at run
25
64
  // time by the `modelRoles` step (the daemon's `model_roles` tool): explicit
26
65
  // `judgeModel` input > repo agentproto.json `models` > daemon config `models`
@@ -37,7 +76,6 @@ const ASK_POLL_MS = 45_000
37
76
 
38
77
  const JUDGE_REF = "@agentproto/session-steward-judge"
39
78
  const VERDICTS = ["done", "abandoned", "blocked", "needs-input", "active"]
40
- const APPLY_VERDICTS = new Set(["done", "abandoned", "blocked", "needs-input"])
41
79
 
42
80
  export const ASK_PROMPT =
43
81
  "Steward check: is your task complete? Reply exactly `STEWARD: DONE <one line>` " +
@@ -61,6 +99,7 @@ function explicitOrRole(explicit, modelRoles, role) {
61
99
  * never a raw `$input.*` that may be absent. */
62
100
  export function resolveSettings(input, modelRoles) {
63
101
  const i = input ?? {}
102
+ const originPolicy = resolveOriginPolicy({ userOrigins: i.userOrigins, closableOrigins: i.closableOrigins })
64
103
  return {
65
104
  idleMinutes: Math.floor(num(i.idleMinutes, DEFAULT_IDLE_MINUTES, { min: 1 })),
66
105
  apply: i.apply === true,
@@ -71,9 +110,34 @@ export function resolveSettings(input, modelRoles) {
71
110
  maxJudged: Math.floor(num(i.maxJudged, DEFAULT_MAX_JUDGED)),
72
111
  askSessions: i.askSessions === true,
73
112
  callerSessionId: typeof i.callerSessionId === "string" && i.callerSessionId ? i.callerSessionId : null,
113
+ callerOrigin: typeof i.callerOrigin === "string" && i.callerOrigin ? i.callerOrigin : null,
114
+ appId: typeof i.appId === "string" && i.appId.trim() ? i.appId.trim() : DEFAULT_APP_ID,
115
+ stableVerdictPasses: Math.floor(num(i.stableVerdictPasses, DEFAULT_STABLE_VERDICT_PASSES, { min: 1 })),
116
+ userOrigins: originPolicy.userOrigins,
117
+ closableOrigins: originPolicy.closableOrigins,
74
118
  }
75
119
  }
76
120
 
121
+ /** The origin policy a settings object carries, as `decideAction` wants it. */
122
+ function policyOf(settings) {
123
+ return { userOrigins: settings?.userOrigins, closableOrigins: settings?.closableOrigins }
124
+ }
125
+
126
+ /** `decideAction` over one candidate entry, with the run's policy folded in.
127
+ * Used by both the apply-queue builders and the report, so the action shown
128
+ * and the action executed can never drift. */
129
+ export function decideFor(entry, planClass, verdict, confidence, settings) {
130
+ return decideAction({
131
+ session: entry,
132
+ planClass,
133
+ verdict,
134
+ confidence,
135
+ apply: settings?.apply === true,
136
+ policy: policyOf(settings),
137
+ minConfidence: settings?.minConfidence,
138
+ })
139
+ }
140
+
77
141
  // ── plan → candidates ────────────────────────────────────────────────────
78
142
 
79
143
  /** Split `session_wrapup_plan`'s entries into this run's work lists. `keep`
@@ -97,21 +161,26 @@ export function splitCandidates(planResult, settings) {
97
161
  }
98
162
  }
99
163
 
100
- /** Rules pass: `close` → done, `stuck` → abandoned — only when `apply`. */
164
+ /** Rules pass: `close` → done, `stuck` → abandoned — only when `apply`, and
165
+ * bounded by origin. A user-origin candidate is never closed: it is queued
166
+ * as a `needs-input` FLAG instead, with the "origine utilisateur" reason. */
101
167
  export function buildRuleApplyQueue(candidates, settings) {
102
168
  if (!settings?.apply) return []
103
- return [
104
- ...(candidates?.close ?? []).map(e => ({
105
- sessionId: e.sessionId,
106
- verdict: "done",
107
- note: `steward-rules: ${(e.reasons ?? []).join("; ") || "close class"}`,
108
- })),
109
- ...(candidates?.stuck ?? []).map(e => ({
110
- sessionId: e.sessionId,
111
- verdict: "abandoned",
112
- note: "stuck starting, never ran",
113
- })),
114
- ]
169
+ const queue = []
170
+ const push = (entries, planClass, closeVerdict, closeNote) => {
171
+ for (const e of entries ?? []) {
172
+ const d = decideFor(e, planClass, closeVerdict, 1, settings)
173
+ if (d.action === "skip") continue
174
+ queue.push({
175
+ sessionId: e.sessionId,
176
+ verdict: d.action === "close" ? closeVerdict : "needs-input",
177
+ note: d.action === "close" ? closeNote(e) : d.reason,
178
+ })
179
+ }
180
+ }
181
+ push(candidates?.close, "close", "done", e => `steward-rules: ${(e.reasons ?? []).join("; ") || "close class"}`)
182
+ push(candidates?.stuck, "stuck", "abandoned", () => "stuck starting, never ran")
183
+ return queue
115
184
  }
116
185
 
117
186
  // ── evidence ─────────────────────────────────────────────────────────────
@@ -128,9 +197,14 @@ function cut(text, max) {
128
197
 
129
198
  /** Plan entry + `session_evidence` → the compact object the judge sees,
130
199
  * under {@link EVIDENCE_MAX_CHARS} once serialized (oldest turns dropped
131
- * first, then the tail signal shortened). */
132
- export function composeEvidence(entry, raw) {
200
+ * first, then the tail signal shortened). `memory` (a folded verdict map
201
+ * from `app_state`, optional) adds the previous verdict for this session.
202
+ * Every field added by the PR-3 enrichment is copied through only when the
203
+ * `session_evidence` tool supplied it — an old daemon still yields the old
204
+ * shape. */
205
+ export function composeEvidence(entry, raw, memory) {
133
206
  const signals = entry?.signals ?? {}
207
+ const previous = memory instanceof Map ? memory.get(entry.sessionId) : memory?.[entry?.sessionId]
134
208
  const evidence = {
135
209
  sessionId: entry.sessionId,
136
210
  label: raw?.label ?? entry.label,
@@ -142,6 +216,8 @@ export function composeEvidence(entry, raw) {
142
216
  busy: raw?.busy === true,
143
217
  rssMB: mb(entry.rssBytes),
144
218
  planReasons: entry.reasons ?? [],
219
+ origin: raw?.origin ?? entry.origin,
220
+ parentSessionId: raw?.parentSessionId ?? entry.parentSessionId,
145
221
  signals: {
146
222
  lastAssistantTail: cut(signals.lastAssistantTail, 800),
147
223
  pendingToolCall: signals.pendingToolCall === true,
@@ -150,6 +226,22 @@ export function composeEvidence(entry, raw) {
150
226
  },
151
227
  worktree: raw?.worktree ?? null,
152
228
  turns: Array.isArray(raw?.turns) ? [...raw.turns] : [],
229
+ ...(raw?.liveChildren !== undefined ? { liveChildren: raw.liveChildren } : {}),
230
+ ...(raw?.continuedFrom ? { continuedFrom: raw.continuedFrom } : {}),
231
+ ...(raw?.continuedTo ? { continuedTo: raw.continuedTo } : {}),
232
+ ...(raw?.tokensIn !== undefined ? { tokensIn: raw.tokensIn } : {}),
233
+ ...(raw?.tokensOut !== undefined ? { tokensOut: raw.tokensOut } : {}),
234
+ ...(raw?.lastTurnErroredAt ? { lastTurnErroredAt: raw.lastTurnErroredAt } : {}),
235
+ ...(raw?.lastTurnError ? { lastTurnError: raw.lastTurnError } : {}),
236
+ ...(raw?.outcome ? { outcome: raw.outcome } : {}),
237
+ ...(raw?.pullRequests ? { pullRequests: raw.pullRequests } : {}),
238
+ ...(raw?.toolStats ? { toolStats: raw.toolStats } : {}),
239
+ ...(raw?.lastToolCall ? { lastToolCall: raw.lastToolCall } : {}),
240
+ ...(raw?.minutesSinceUserMessage !== undefined ? { minutesSinceUserMessage: raw.minutesSinceUserMessage } : {}),
241
+ ...(raw?.minutesSinceAgentMessage !== undefined ? { minutesSinceAgentMessage: raw.minutesSinceAgentMessage } : {}),
242
+ ...(previous
243
+ ? { previousVerdict: { verdict: previous.verdict, confidence: previous.confidence ?? null, streak: previous.streak ?? 1, ts: previous.ts ?? null } }
244
+ : {}),
153
245
  }
154
246
  while (JSON.stringify(evidence).length > EVIDENCE_MAX_CHARS && evidence.turns.length > 0) evidence.turns.shift()
155
247
  if (JSON.stringify(evidence).length > EVIDENCE_MAX_CHARS) evidence.signals.lastAssistantTail = cut(evidence.signals.lastAssistantTail, 200)
@@ -160,15 +252,22 @@ export function buildJudgePrompt(evidence) {
160
252
  return (
161
253
  "You are the session steward's judge. Decide whether ONE idle AI coding-agent session " +
162
254
  "is finished, from the evidence below. Do NOT call any tool — answer from the evidence alone.\n\n" +
163
- "Verdicts:\n" +
164
- "- `done`: the task visibly finished — a PR was opened or merged, a final report was given, " +
165
- "or the user said thanks/ok with nothing pending.\n" +
166
- "- `abandoned`: superseded or a dead end, with nothing worth keeping.\n" +
167
- "- `blocked`: waiting on something external (CI, another session, a dependency).\n" +
168
- "- `needs-input`: waiting on a human answer or decision.\n" +
169
- "- `active`: mid-work — keep it.\n" +
170
- "When unsure, say `active` with a LOW confidence. Closing a session that still had work " +
171
- "is worse than leaving an idle one open.\n\n" +
255
+ "Verdicts (concrete signals — see the evidence fields named in each):\n" +
256
+ "- `done`: finished with nothing pending — `pullRequests.merged` > 0 or " +
257
+ "`worktree.pr.state`=\"merged\"; or `pullRequests.opened` > 0 with a final report and no " +
258
+ "open question; or the last tool call is a `message_parent` with `kind:\"done\"`; or " +
259
+ "`outcome.verdict`=\"done\"; or the user's last message is an acknowledgement with no " +
260
+ "pending question.\n" +
261
+ "- `abandoned`: superseded or a dead end — `outcome.verdict`=\"abandoned\"/\"failed\", or " +
262
+ "the worktree is gone/merged elsewhere with no open PR and no pending question.\n" +
263
+ "- `blocked`: waiting on something EXTERNAL — an open PR with CI/review pending, " +
264
+ "`liveChildren` > 0, or a `lastTurnError` that clears on its own.\n" +
265
+ "- `needs-input`: waiting on a HUMAN — `awaitingInput` true, or the LAST assistant turn " +
266
+ "ends in a question to the user/operator.\n" +
267
+ "- `active`: mid-work — `busy`, a progress update with no conclusion, recent distinct " +
268
+ "`toolStats`, or an unchanged `previousVerdict` of active.\n" +
269
+ "When the evidence is thin or ambiguous, say `active` with a LOW confidence. Closing a " +
270
+ "session that still had work is worse than leaving an idle one open.\n\n" +
172
271
  "Reply with ONLY one JSON object, no prose, no code fence:\n" +
173
272
  `{"sessionId": "${evidence.sessionId}", "verdict": "done"|"abandoned"|"blocked"|"needs-input"|"active", ` +
174
273
  '"confidence": <number 0..1>, "reason": "<one line>"}\n\n' +
@@ -184,7 +283,7 @@ function foldEvidence(b) {
184
283
  if (!raw || raw.sessionId !== b.item?.sessionId) {
185
284
  throw new Error(`session_evidence answered for '${raw?.sessionId}', expected '${b.item?.sessionId}'`)
186
285
  }
187
- const evidence = composeEvidence(b.item, raw)
286
+ const evidence = composeEvidence(b.item, raw, b.steps.memory)
188
287
  return { entry: b.item, evidence, judgePrompt: buildJudgePrompt(evidence) }
189
288
  }
190
289
 
@@ -380,18 +479,25 @@ export function mergeDeclared(verdicts, askQueue, askResult) {
380
479
  // ── judged apply ─────────────────────────────────────────────────────────
381
480
 
382
481
  /** `done`/`abandoned` (close) and `blocked`/`needs-input` (flag) at or above
383
- * `minConfidence` — only when `apply`. A malformed reply is `active`/0 and
384
- * can never qualify. */
482
+ * `minConfidence` — only when `apply`, and bounded by origin: a user-origin
483
+ * candidate is downgraded to a `needs-input` FLAG, never a close. A malformed
484
+ * reply is `active`/0 and can never qualify. */
385
485
  export function buildJudgedApplyQueue(finalVerdicts, settings) {
386
486
  if (!settings?.apply) return []
387
- return (finalVerdicts ?? [])
388
- .filter(r => !r.malformed && APPLY_VERDICTS.has(r.verdict) && r.confidence >= settings.minConfidence)
389
- .map(r => ({
487
+ const queue = []
488
+ for (const r of finalVerdicts ?? []) {
489
+ if (r.malformed) continue
490
+ const d = decideFor(r.entry, "judge", r.verdict, r.confidence, settings)
491
+ if (d.action === "skip") continue
492
+ const isFlagVerdict = r.verdict === "blocked" || r.verdict === "needs-input"
493
+ queue.push({
390
494
  sessionId: r.entry.sessionId,
391
- verdict: r.verdict,
495
+ verdict: d.action === "close" ? r.verdict : isFlagVerdict ? r.verdict : "needs-input",
392
496
  judgedBy: r.source === "declared" ? `steward-ask:${r.entry.sessionId}` : r.judgedBy ?? r.judgeSessionId ?? "steward-judge",
393
- note: r.reason,
394
- }))
497
+ note: d.action === "close" ? r.reason : `${d.reason}${r.reason ? ` — ${r.reason}` : ""}`,
498
+ })
499
+ }
500
+ return queue
395
501
  }
396
502
 
397
503
  // ── report ───────────────────────────────────────────────────────────────
@@ -418,15 +524,33 @@ function cell(s) {
418
524
  return String(s ?? "").replace(/\|/g, "\\|").replace(/\n/g, " ")
419
525
  }
420
526
 
421
- function actionOf(applied, apply, wouldAct) {
422
- if (applied) return applied.ok ? applied.action ?? "applied" : `refused (${applied.error})`
423
- if (!apply) return wouldAct ? "none (dry run)" : "none"
424
- return "none"
527
+ function actionOf(applied) {
528
+ return applied.ok ? applied.action ?? "applied" : `refused (${applied.error})`
529
+ }
530
+
531
+ /** The `origin` column: the provenance label plus `(user)` when the origin
532
+ * policy bounds this candidate to flag-only. `(none, user)` is a root with
533
+ * no origin and no parent — human-launched, never closed. */
534
+ function originCell(entry, settings) {
535
+ const origin = entry?.origin
536
+ const userBound = classifyOrigin(entry, policyOf(settings)) === "user"
537
+ if (!origin) return userBound ? "(none, user)" : "(none)"
538
+ return userBound ? `${origin} (user)` : origin
539
+ }
540
+
541
+ /** The `action` column: what actually happened (apply) or the retained action
542
+ * the policy decided (dry run / skipped). When an outcome exists, the
543
+ * retained-action label stays visible next to it — that label is what
544
+ * carries the origin bound ("flag (origine utilisateur)"). */
545
+ function actionCell(id, decision, applied) {
546
+ const a = applied.get(id)
547
+ if (!a) return decision.reason
548
+ return `${actionOf(a)} — ${decision.reason}`
425
549
  }
426
550
 
427
551
  export function buildReport(b) {
428
552
  const s = b.steps.settings ?? resolveSettings(b.input)
429
- const c = b.steps.candidates ?? { close: [], stuck: [], judge: [], judgeOverflow: [] }
553
+ const c = b.steps.candidatesPlus ?? b.steps.candidates ?? { close: [], stuck: [], judge: [], judgeOverflow: [] }
430
554
  const verdicts = b.steps.finalVerdicts ?? []
431
555
  const applied = collectApplyResults(b.steps.autoApply, b.steps.judgedApply)
432
556
  const lines = []
@@ -437,27 +561,31 @@ export function buildReport(b) {
437
561
  `(jev \`${s.jevModel}\`, agent \`${s.judgeModel ?? "agent default"}\`)` +
438
562
  (s.askSessions ? " · askSessions on" : ""),
439
563
  )
564
+ lines.push(`origins: user=${s.userOrigins.join(", ")} · closable=${s.closableOrigins.join(", ")}`)
565
+ // Host saturation header first, report-only (mission item 9).
566
+ for (const line of saturationHeader(b.steps.hostLoad)) lines.push(line)
440
567
  if (!s.apply) lines.push("", "_Dry run: nothing was closed or flagged. Re-run with `apply: true` to act._")
441
568
  lines.push("")
442
- lines.push("| class | session | idle | RAM | verdict | confidence | reason | action |")
443
- lines.push("|---|---|---|---|---|---|---|---|")
569
+ lines.push("| class | session | origin | idle | RAM | verdict | confidence | reason | action |")
570
+ lines.push("|---|---|---|---|---|---|---|---|---|")
444
571
  const row = (cls, e, verdict, conf, reason, action) =>
445
572
  lines.push(
446
- `| ${cls} | ${cell(e.label ?? e.sessionId)} | ${e.idleMinutes ?? "?"} min | ${fmtMB(e.rssBytes)} | ` +
573
+ `| ${cls} | ${cell(e.label ?? e.sessionId)} | ${cell(originCell(e, s))} | ${e.idleMinutes ?? "?"} min | ${fmtMB(e.rssBytes)} | ` +
447
574
  `${cell(verdict)} | ${conf === undefined ? "—" : conf.toFixed(2)} | ${cell(reason)} | ${cell(action)} |`,
448
575
  )
449
- for (const e of c.close) row("close", e, "done (rules)", undefined, (e.reasons ?? []).join("; "), actionOf(applied.get(e.sessionId), s.apply, true))
450
- for (const e of c.stuck) row("stuck", e, "abandoned (rules)", undefined, "stuck starting, never ran", actionOf(applied.get(e.sessionId), s.apply, true))
576
+ for (const e of c.close) {
577
+ const d = decideFor(e, "close", "done", 1, s)
578
+ row("close", e, "done (rules)", undefined, (e.reasons ?? []).join("; "), actionCell(e.sessionId, d, applied))
579
+ }
580
+ for (const e of c.stuck) {
581
+ const d = decideFor(e, "stuck", "abandoned", 1, s)
582
+ row("stuck", e, "abandoned (rules)", undefined, (e.reasons ?? []).join("; ") || "stuck starting, never ran", actionCell(e.sessionId, d, applied))
583
+ }
451
584
  for (const r of verdicts) {
452
- const wouldAct = !r.malformed && APPLY_VERDICTS.has(r.verdict) && r.confidence >= s.minConfidence
453
- const action = applied.get(r.entry.sessionId)
454
- ? actionOf(applied.get(r.entry.sessionId), s.apply, wouldAct)
455
- : wouldAct
456
- ? s.apply ? "none" : "none (dry run)"
457
- : "untouched (below threshold or active)"
585
+ const d = decideFor(r.entry, "judge", r.verdict, r.confidence, s)
458
586
  const by = r.source === "declared" ? " (declared)" : r.source === "jev" ? " (jev)" : r.source === "judged" ? " (agent)" : ""
459
587
  const reason = r.jevFallback ? `${r.reason} [jev failed: ${r.jevFallback} → agent judge]` : r.reason
460
- row("judge", r.entry, `${r.verdict}${by}`, r.confidence, reason, action)
588
+ row("judge", r.entry, `${r.verdict}${by}`, r.confidence, reason, actionCell(r.entry.sessionId, d, applied))
461
589
  }
462
590
  for (const e of c.judgeOverflow ?? []) row("judge", e, "—", undefined, `not judged this run (maxJudged ${s.maxJudged})`, "none")
463
591
  lines.push("")
@@ -489,9 +617,263 @@ export function buildReport(b) {
489
617
  }
490
618
  lines.push(`- RAM freed (closed sessions): ${fmtMB(freed)}`)
491
619
  lines.push(`- RAM still held by idle sessions: ${fmtMB(held)}`)
620
+
621
+ // Explicit "0 candidates" explanation (mission item 8) — say WHY, instead
622
+ // of leaving an empty table to interpret.
623
+ const scan = b.steps.scan
624
+ if (scan && c.close.length === 0 && c.stuck.length === 0 && verdicts.length === 0) {
625
+ lines.push(`- ${explainZeroCandidates(scan.counts)}`)
626
+ }
627
+
628
+ // Nudge proposals (mission items 1-2) — report only, NEVER executed here.
629
+ const prop = b.steps.proposals ?? { proposals: [], observed: [] }
630
+ if ((prop.proposals ?? []).length > 0 || (prop.observed ?? []).length > 0) {
631
+ lines.push("", "## Proposals (report only — no nudge is sent by this workflow)")
632
+ for (const p of prop.proposals ?? []) lines.push(`- ${p.kind} nudge → ${p.sessionId} — ${p.reason}`)
633
+ for (const p of prop.observed ?? []) lines.push(`- observed → ${p.sessionId} — ${p.reason} (no nudge: ${p.suppressed})`)
634
+ }
635
+
636
+ // Terminal sessions with no outcome (mission item 5).
637
+ const relabel = b.steps.relabelQueue ?? []
638
+ if (relabel.length > 0) {
639
+ lines.push("", "## Terminal sessions missing an outcome (relabel candidates)")
640
+ for (const r of relabel) lines.push(`- ${r.sessionId} → ${r.proposedVerdict} — ${r.reason}`)
641
+ }
642
+
643
+ // Verdict memory / cache (mission item 10).
644
+ const memory = b.steps.memory
645
+ const cachedCount = verdicts.filter(r => r.source === "cache").length
646
+ if (memory instanceof Map && memory.size > 0) {
647
+ lines.push(`- verdict memory: ${memory.size} session(s) known` + (cachedCount > 0 ? `, ${cachedCount} served from cache` : ""))
648
+ }
649
+
492
650
  return lines.join("\n")
493
651
  }
494
652
 
653
+ // ── live scan, loop/stall, memory (mission items 1-10) ───────────────────
654
+
655
+ const TERMINAL_STATUSES = new Set(["killed", "exited", "error", "stopped", "completed", "failed"])
656
+
657
+ function idleMinutesOf(row, nowMs) {
658
+ const ts = row?.lastActivityAt ?? row?.startedAt
659
+ const ms = ts ? Date.parse(ts) : Number.NaN
660
+ return Number.isFinite(ms) ? Math.max(0, (nowMs - ms) / 60_000) : 0
661
+ }
662
+
663
+ function liveRowsOf(liveSessions) {
664
+ if (Array.isArray(liveSessions?.items)) return liveSessions.items
665
+ if (Array.isArray(liveSessions)) return liveSessions
666
+ return []
667
+ }
668
+
669
+ /**
670
+ * One deterministic pass over the live `session_list` rows: busy sessions
671
+ * (loop/stall scan), idle sessions (zero-candidate accounting), terminal
672
+ * sessions missing an outcome (relabel candidates), never-ran 0/0 sessions,
673
+ * and everything excluded (self / same cron job / archived / pinned / pty /
674
+ * keepAlive). Pure over the rows + settings + an injected `nowMs`.
675
+ */
676
+ export function scanLive(liveSessions, settings, nowMs) {
677
+ const rows = liveRowsOf(liveSessions)
678
+ const policy = policyOf(settings)
679
+ const self = settings?.callerSessionId ?? null
680
+ const idleThreshold = settings?.idleMinutes ?? DEFAULT_IDLE_MINUTES
681
+ const busy = []
682
+ const idle = []
683
+ const terminal = []
684
+ const terminalRelabel = []
685
+ const neverRan = []
686
+ const excluded = []
687
+ const loopQueue = []
688
+ const stallInputs = []
689
+ let liveCount = 0
690
+ for (const s of rows) {
691
+ const id = s?.id
692
+ if (!id) continue
693
+ if (self && id === self) {
694
+ excluded.push({ sessionId: id, reason: "caller session" })
695
+ continue
696
+ }
697
+ if (isSelfExcluded(s, { callerSessionId: self, callerOrigin: settings?.callerOrigin }).excluded) {
698
+ excluded.push({ sessionId: id, reason: "same cron job as caller" })
699
+ continue
700
+ }
701
+ if (s.archived === true) {
702
+ excluded.push({ sessionId: id, reason: "archived" })
703
+ continue
704
+ }
705
+ if (s.pinned === true) {
706
+ excluded.push({ sessionId: id, reason: "pinned" })
707
+ continue
708
+ }
709
+ if (s.pty === true) {
710
+ excluded.push({ sessionId: id, reason: "pty" })
711
+ continue
712
+ }
713
+ const originClass = classifyOrigin(s, policy)
714
+ const label = s.label ?? s.name
715
+ const idleMinutes = idleMinutesOf(s, nowMs)
716
+ if (TERMINAL_STATUSES.has(String(s.status ?? ""))) {
717
+ terminal.push({ sessionId: id, origin: s.origin, originClass, label })
718
+ const cand = terminalRelabelCandidate(s)
719
+ if (cand.candidate) {
720
+ terminalRelabel.push({ sessionId: id, origin: s.origin, originClass, label, proposedVerdict: cand.proposedVerdict, reason: cand.reason })
721
+ }
722
+ continue
723
+ }
724
+ if (s.status !== "running" && s.status !== "starting") continue
725
+ liveCount++
726
+ if (s.keepAlive === true) {
727
+ excluded.push({ sessionId: id, reason: "keepAlive" })
728
+ continue
729
+ }
730
+ const row = { sessionId: id, origin: s.origin, originClass, label, idleMinutes, lastTurnErroredAt: s.lastTurnErroredAt ?? null }
731
+ if (isNeverRan(s)) neverRan.push(row)
732
+ if (s.busy === true) {
733
+ busy.push(row)
734
+ loopQueue.push({ sessionId: id, originClass, label })
735
+ stallInputs.push({ sessionId: id, originClass, busy: true, idleMinutes, lastTurnErroredAt: s.lastTurnErroredAt ?? null })
736
+ } else if (idleMinutes >= idleThreshold) {
737
+ idle.push(row)
738
+ }
739
+ }
740
+ const counts = {
741
+ live: liveCount,
742
+ busy: busy.length,
743
+ idle: idle.length,
744
+ terminal: terminal.length,
745
+ terminalRelabel: terminalRelabel.length,
746
+ neverRan: neverRan.length,
747
+ excluded: excluded.length,
748
+ }
749
+ return { busy, idle, terminal, terminalRelabel, neverRan, excluded, loopQueue, stallInputs, counts }
750
+ }
751
+
752
+ /** Fold the never-ran 0/0 sessions into the plan as `stuck` (no judge,
753
+ * whatever the idle) and drop them from the judge queue — mission item 3. */
754
+ export function mergeNeverRan(candidates, scan) {
755
+ const never = scan?.neverRan ?? []
756
+ const neverIds = new Set(never.map(n => n.sessionId))
757
+ const existing = new Set((candidates?.stuck ?? []).map(e => e.sessionId))
758
+ const added = never
759
+ .filter(n => !existing.has(n.sessionId))
760
+ .map(n => ({
761
+ sessionId: n.sessionId,
762
+ ...(n.label ? { label: n.label } : {}),
763
+ idleMinutes: Math.round(n.idleMinutes ?? 0),
764
+ class: "stuck",
765
+ reasons: ["0 tokens in/out — never ran"],
766
+ signals: {},
767
+ ...(n.origin ? { origin: n.origin } : {}),
768
+ }))
769
+ return {
770
+ ...candidates,
771
+ stuck: [...(candidates?.stuck ?? []), ...added],
772
+ judge: (candidates?.judge ?? []).filter(e => !neverIds.has(e.sessionId)),
773
+ judgeOverflow: (candidates?.judgeOverflow ?? []).filter(e => !neverIds.has(e.sessionId)),
774
+ }
775
+ }
776
+
777
+ /** `tool_calls_list` map item → the loop verdict + stats for one session. */
778
+ export function analyzeLoopItem(b) {
779
+ const item = b.item ?? {}
780
+ const raw = b.steps.loopCallsOne
781
+ const records = Array.isArray(raw?.records) ? raw.records : Array.isArray(raw) ? raw : []
782
+ const r = detectLoop(records, { nowMs: Date.now() })
783
+ return { sessionId: item.sessionId, label: item.label, originClass: item.originClass, ...r }
784
+ }
785
+
786
+ /** Stall verdicts for every busy live session. */
787
+ export function analyzeStalls(scan, nowMs) {
788
+ return (scan?.stallInputs ?? []).map(s => ({
789
+ sessionId: s.sessionId,
790
+ originClass: s.originClass,
791
+ ...detectStall({ busy: s.busy, idleMinutes: s.idleMinutes, lastTurnErroredAt: s.lastTurnErroredAt, nowMs }),
792
+ }))
793
+ }
794
+
795
+ /** Fold the `app_state` read into the per-session verdict memory map. */
796
+ export function foldMemory(b) {
797
+ const events = settled(b.steps.memoryRead).ok.flatMap(r => (Array.isArray(r.value?.events) ? r.value.events : []))
798
+ return foldVerdictMemory(events)
799
+ }
800
+
801
+ /** Judge candidates minus those already judged the same verdict on the same
802
+ * evidence fingerprint for `stableVerdictPasses` passes (the cache). */
803
+ export function buildJudgeQueueFiltered(evidenceResult, memory, settings) {
804
+ const rows = settled(evidenceResult).ok.map(r => r.value)
805
+ const queue = []
806
+ const cached = []
807
+ for (const q of rows) {
808
+ const fingerprint = evidenceFingerprint(q.evidence)
809
+ const decision = shouldRejudge(memory, q.entry.sessionId, fingerprint, { stablePasses: settings?.stableVerdictPasses })
810
+ if (!decision.rejudge && decision.cached) cached.push({ ...q, cached: decision.cached, fingerprint })
811
+ else queue.push(q)
812
+ }
813
+ return { queue, cached }
814
+ }
815
+
816
+ /** Cached rows as verdict rows, so they appear in the report and (when they
817
+ * carry a confident close verdict) can still be applied without re-judging. */
818
+ export function buildCachedVerdicts(cachedQueue) {
819
+ return (cachedQueue ?? []).map(q => ({
820
+ entry: q.entry,
821
+ evidence: q.evidence,
822
+ verdict: q.cached.verdict,
823
+ confidence: typeof q.cached.confidence === "number" ? q.cached.confidence : 0,
824
+ reason: `cached verdict (stable ${q.cached.streak ?? "?"} passes, evidence unchanged)`,
825
+ source: "cache",
826
+ cached: true,
827
+ judgedBy: q.cached.judgedBy ?? "steward-cache",
828
+ }))
829
+ }
830
+
831
+ /** `collectVerdicts` + the cached rows (cache rows are never re-judged). */
832
+ export function buildVerdicts(b) {
833
+ const jq = b.steps.judgeQueue ?? {}
834
+ return [
835
+ ...collectVerdicts(b.steps.evidence, jq.queue, b.steps.jevJudge, b.steps.jevQueue, b.steps.agentJudgeQueue, b.steps.judge),
836
+ ...buildCachedVerdicts(jq.cached),
837
+ ]
838
+ }
839
+
840
+ /** The report's nudge PROPOSALS (loop → interrupt, stall → continue) plus
841
+ * the user-origin findings reported as observed-only. Never a close. */
842
+ export function buildProposalsStep(scan, loopResults, settings, nowMs) {
843
+ const stalls = analyzeStalls(scan, nowMs)
844
+ const { proposals, observed } = buildProposals({ loopResults, stallResults: stalls })
845
+ return { proposals, observed, stalls }
846
+ }
847
+
848
+ /** Terminal sessions missing an outcome, as relabel candidates. */
849
+ export function buildRelabelQueue(scan) {
850
+ return (scan?.terminalRelabel ?? []).map(t => ({ ...t }))
851
+ }
852
+
853
+ /** The `app_state` events to append for this pass's verdicts. The memory is
854
+ * written on every pass (it is a ledger, never a session action) so streaks
855
+ * accumulate and the cache can engage. */
856
+ export function buildMemoryWriteQueue(finalVerdicts, settings) {
857
+ if (!settings?.appId) return []
858
+ const out = []
859
+ for (const r of finalVerdicts ?? []) {
860
+ if (!r?.entry?.sessionId || r.malformed) continue
861
+ const fingerprint = r.evidence ? evidenceFingerprint(r.evidence) : null
862
+ out.push({
863
+ appId: settings.appId,
864
+ event: verdictMemoryEvent({
865
+ sessionId: r.entry.sessionId,
866
+ verdict: r.verdict,
867
+ confidence: r.confidence,
868
+ fingerprint,
869
+ judgedBy: r.judgedBy ?? r.source ?? null,
870
+ note: r.reason,
871
+ }),
872
+ })
873
+ }
874
+ return out
875
+ }
876
+
495
877
  // ── the workflow ─────────────────────────────────────────────────────────
496
878
 
497
879
  export default {
@@ -502,6 +884,8 @@ export default {
502
884
  "`close`/`stuck` sessions, judge the ambiguous `judge` ones with a cheap " +
503
885
  "one-shot model over compact evidence, optionally ask a session directly, " +
504
886
  "then close or flag the confident verdicts with a recorded outcome — and report. " +
887
+ "Origin-bounded: a human-launched session (`chat-starter`, `vscode`, or a " +
888
+ "root with no origin and no parent) is only ever flagged, never closed. " +
505
889
  "Dry run unless `apply` is true.",
506
890
  version: "0.1.0",
507
891
  inputs: {
@@ -514,6 +898,11 @@ export default {
514
898
  maxJudged: { type: "number", description: `Most \`judge\` sessions judged per run, most RAM first. Default ${DEFAULT_MAX_JUDGED}.`, default: DEFAULT_MAX_JUDGED },
515
899
  askSessions: { type: "boolean", description: "Ask low-confidence idle sessions directly whether they're done. Default false — it spends a turn in someone else's conversation.", default: false },
516
900
  callerSessionId: { type: "string", description: "The calling session's id — never a candidate. The CLI passes AGENTPROTO_SESSION_ID." },
901
+ callerOrigin: { type: "string", description: "The calling session's origin (`cron:<jobId>`) — an older run of the SAME cron job is never judged as user work." },
902
+ appId: { type: "string", description: `Installed app whose \`app_state\` ledger holds the verdict memory. Default ${DEFAULT_APP_ID}.` },
903
+ stableVerdictPasses: { type: "number", description: `Consecutive passes on an unchanged evidence fingerprint before the judge cache stops re-judging. Default ${DEFAULT_STABLE_VERDICT_PASSES}.`, default: DEFAULT_STABLE_VERDICT_PASSES },
904
+ userOrigins: { type: "array", description: `Origins that are ALWAYS flag-only, never closed (a human is in the loop). Trailing \`*\` is a prefix wildcard. Default ${JSON.stringify(DEFAULT_USER_ORIGINS)}.`, items: { type: "string" }, default: DEFAULT_USER_ORIGINS },
905
+ closableOrigins: { type: "array", description: `Origins that may be closed under the current rules (cron jobs, gates). Trailing \`*\` is a prefix wildcard. Executors (a session with a parentSessionId) are closable regardless. Default ${JSON.stringify(DEFAULT_CLOSABLE_ORIGINS)}.`, items: { type: "string" }, default: DEFAULT_CLOSABLE_ORIGINS },
517
906
  },
518
907
  outputs: {},
519
908
  steps: [
@@ -531,7 +920,14 @@ export default {
531
920
  inputs: { idleMinutes: "$steps.settings.idleMinutes" },
532
921
  },
533
922
  { id: "candidates", kind: "transform", compute: b => splitCandidates(b.steps.plan, b.steps.settings) },
534
- { id: "ruleApplyQueue", kind: "transform", compute: b => buildRuleApplyQueue(b.steps.candidates, b.steps.settings) },
923
+ // Host saturation header (report only — mission item 9) and the live
924
+ // session scan behind loop/stall/never-ran/terminal rules (items 1-5).
925
+ { id: "hostLoad", kind: "tool", tool: "host_load", inputs: {} },
926
+ { id: "liveSessions", kind: "tool", tool: "session_list", inputs: { full: true } },
927
+ { id: "scan", kind: "transform", compute: b => scanLive(b.steps.liveSessions, b.steps.settings, Date.now()) },
928
+ // Never-ran 0/0 sessions are `stuck` immediately, never judged (item 3).
929
+ { id: "candidatesPlus", kind: "transform", compute: b => mergeNeverRan(b.steps.candidates, b.steps.scan) },
930
+ { id: "ruleApplyQueue", kind: "transform", compute: b => buildRuleApplyQueue(b.steps.candidatesPlus, b.steps.settings) },
535
931
  {
536
932
  // Empty unless `apply` — a dry run dispatches no apply call at all.
537
933
  id: "autoApply",
@@ -548,10 +944,46 @@ export default {
548
944
  },
549
945
  ],
550
946
  },
947
+ // Verdict memory (item 10): read the app_state ledger best-effort. The
948
+ // map is empty when no app id is set, so a caller can turn memory off.
949
+ { id: "memoryQueue", kind: "transform", compute: b => (b.steps.settings?.appId ? [{ appId: b.steps.settings.appId }] : []) },
950
+ {
951
+ id: "memoryRead",
952
+ kind: "map",
953
+ over: "$steps.memoryQueue",
954
+ parallelism: 1,
955
+ onError: "collect",
956
+ steps: [
957
+ {
958
+ id: "memoryReadOne",
959
+ kind: "tool",
960
+ tool: "app_state_list",
961
+ inputs: { appId: "$item.appId", stage: "session-steward", kinds: ["note"], limit: 500 },
962
+ },
963
+ ],
964
+ },
965
+ { id: "memory", kind: "transform", compute: foldMemory },
966
+ // Loop sanity over the busy sessions (item 1) — one tool_calls_list each.
967
+ {
968
+ id: "loopScan",
969
+ kind: "map",
970
+ over: "$steps.scan.loopQueue",
971
+ parallelism: 4,
972
+ onError: "collect",
973
+ steps: [
974
+ { id: "loopCallsOne", kind: "tool", tool: "tool_calls_list", inputs: { sessionId: "$item.sessionId", lastN: 60 } },
975
+ { id: "loopFold", kind: "transform", compute: analyzeLoopItem },
976
+ ],
977
+ },
978
+ { id: "loopResults", kind: "transform", compute: b => settled(b.steps.loopScan).ok.map(r => r.value) },
979
+ // Nudge PROPOSALS (never executed here): loop → interrupt, stall →
980
+ // continue, at most one per session per pass, user origins observed only.
981
+ { id: "proposals", kind: "transform", compute: b => buildProposalsStep(b.steps.scan, b.steps.loopResults, b.steps.settings, Date.now()) },
982
+ { id: "relabelQueue", kind: "transform", compute: b => buildRelabelQueue(b.steps.scan) },
551
983
  {
552
984
  id: "evidence",
553
985
  kind: "map",
554
- over: "$steps.candidates.judge",
986
+ over: "$steps.candidatesPlus.judge",
555
987
  parallelism: 4,
556
988
  onError: "collect",
557
989
  steps: [
@@ -561,11 +993,12 @@ export default {
561
993
  { id: "evidenceFold", kind: "transform", compute: foldEvidence },
562
994
  ],
563
995
  },
564
- { id: "judgeQueue", kind: "transform", compute: b => settled(b.steps.evidence).ok.map(r => r.value) },
996
+ // Judge queue minus sessions cached by stable verdict+fingerprint (item 10).
997
+ { id: "judgeQueue", kind: "transform", compute: b => buildJudgeQueueFiltered(b.steps.evidence, b.steps.memory, b.steps.settings) },
565
998
  {
566
999
  id: "jevQueue",
567
1000
  kind: "transform",
568
- compute: b => (b.steps.settings?.judge === "agent" ? [] : b.steps.judgeQueue ?? []),
1001
+ compute: b => (b.steps.settings?.judge === "agent" ? [] : b.steps.judgeQueue?.queue ?? []),
569
1002
  },
570
1003
  {
571
1004
  // Jev backend: one calibrated `choice` call per candidate. Never an
@@ -588,7 +1021,7 @@ export default {
588
1021
  {
589
1022
  id: "agentJudgeQueue",
590
1023
  kind: "transform",
591
- compute: b => buildAgentJudgeQueue(b.steps.judgeQueue, b.steps.jevQueue, b.steps.jevJudge, b.steps.settings),
1024
+ compute: b => buildAgentJudgeQueue(b.steps.judgeQueue?.queue, b.steps.jevQueue, b.steps.jevJudge, b.steps.settings),
592
1025
  },
593
1026
  {
594
1027
  // One-shot judge per candidate. The engine releases (kills + archives)
@@ -617,9 +1050,7 @@ export default {
617
1050
  },
618
1051
  ],
619
1052
  },
620
- { id: "verdicts", kind: "transform", compute: b =>
621
- collectVerdicts(b.steps.evidence, b.steps.judgeQueue, b.steps.jevJudge, b.steps.jevQueue, b.steps.agentJudgeQueue, b.steps.judge),
622
- },
1053
+ { id: "verdicts", kind: "transform", compute: buildVerdicts },
623
1054
  { id: "askQueue", kind: "transform", compute: b => buildAskQueue(b.steps.verdicts, b.steps.settings) },
624
1055
  {
625
1056
  // Empty unless `askSessions`. ONE prompt per session (queue:false — a
@@ -689,14 +1120,30 @@ export default {
689
1120
  },
690
1121
  ],
691
1122
  },
1123
+ // Verdict memory write-back (item 10) — a ledger append, never a session
1124
+ // action; best-effort (an uninstalled app just yields no memory).
1125
+ { id: "memoryWriteQueue", kind: "transform", compute: b => buildMemoryWriteQueue(b.steps.finalVerdicts, b.steps.settings) },
1126
+ {
1127
+ id: "memoryWrite",
1128
+ kind: "map",
1129
+ over: "$steps.memoryWriteQueue",
1130
+ parallelism: 1,
1131
+ onError: "collect",
1132
+ steps: [
1133
+ { id: "memoryWriteOne", kind: "tool", tool: "app_state_append", inputs: { appId: "$item.appId", event: "$item.event" } },
1134
+ ],
1135
+ },
692
1136
  { id: "report", kind: "transform", compute: b => buildReport(b) },
693
1137
  ],
694
1138
  result: {
695
1139
  report: "$steps.report",
696
1140
  apply: "$steps.settings.apply",
697
- candidates: "$steps.candidates",
1141
+ candidates: "$steps.candidatesPlus",
698
1142
  verdicts: "$steps.finalVerdicts",
699
1143
  autoApply: "$steps.autoApply",
700
1144
  judgedApply: "$steps.judgedApply",
1145
+ proposals: "$steps.proposals",
1146
+ relabel: "$steps.relabelQueue",
1147
+ scan: "$steps.scan",
701
1148
  },
702
1149
  }