@agentproto/apps 0.21.0 → 0.21.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/package.json +2 -2
- package/repo-maintenance/.agentproto/workflows/maintain/entry.mjs +9 -1
- package/session-steward/.agentproto/workflows/session-steward/WORKFLOW.md +36 -6
- package/session-steward/.agentproto/workflows/session-steward/cron-rules.mjs +198 -56
- package/session-steward/.agentproto/workflows/session-steward/entry.mjs +84 -11
- package/session-steward/README.md +30 -3
- package/session-steward/routines/session-steward-hourly/ROUTINE.md +3 -0
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@agentproto/apps",
|
|
3
|
-
"version": "0.21.
|
|
3
|
+
"version": "0.21.2",
|
|
4
4
|
"description": "@agentproto/apps — a home for ready-made agentproto apps (teams of agents + their workflows), declared with @agentproto/app-kit. Import a team (code-team, content-team, …) and use any of its agents/workflows from your own host: in-process via toMastraAgents(), or emit its manifests to disk.",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"agentproto",
|
|
@@ -175,7 +175,7 @@
|
|
|
175
175
|
"zod": "^4.6.5",
|
|
176
176
|
"@agentproto/agent": "0.2.5",
|
|
177
177
|
"@agentproto/app-client": "0.4.1",
|
|
178
|
-
"@agentproto/app-kit": "1.7.
|
|
178
|
+
"@agentproto/app-kit": "1.7.2",
|
|
179
179
|
"@agentproto/workflow": "0.7.1"
|
|
180
180
|
},
|
|
181
181
|
"peerDependencies": {
|
|
@@ -168,13 +168,21 @@ export function reviewableWorktrees(worktreeGcPlanResult) {
|
|
|
168
168
|
const entries = Array.isArray(worktreeGcPlanResult?.plan) ? worktreeGcPlanResult.plan : []
|
|
169
169
|
for (const w of entries) {
|
|
170
170
|
if (w?.class !== "hold" || !w.branch || !w.path) continue
|
|
171
|
-
if (w.liveness?.state !== "idle" || w.tree
|
|
171
|
+
if (w.liveness?.state !== "idle" || treeState(w.tree) !== "clean") continue
|
|
172
172
|
if (w.integration?.state === "open") continue
|
|
173
173
|
out.set(w.path, w.branch)
|
|
174
174
|
}
|
|
175
175
|
return out
|
|
176
176
|
}
|
|
177
177
|
|
|
178
|
+
/** A worktree_gc plan entry's tree state. The daemon's `worktree_gc` tool
|
|
179
|
+
* flattens it to the bare discriminant (`tree: "clean"`, `toGcPlanEntryView`
|
|
180
|
+
* in packages/cli/src/commands/worktree.ts); the engine's own `GcPlanEntry`
|
|
181
|
+
* (and `agentproto worktree gc --json`) carries `{ state, … }`. */
|
|
182
|
+
function treeState(tree) {
|
|
183
|
+
return typeof tree === "string" ? tree : tree?.state
|
|
184
|
+
}
|
|
185
|
+
|
|
178
186
|
/** The worktree path a branch_gc entry is held by, when that hold is the only
|
|
179
187
|
* thing between it and `review`: held for `worktree`, unmerged, old enough,
|
|
180
188
|
* and the worktree is one {@link reviewableWorktrees} kept. Else null. */
|
|
@@ -56,6 +56,24 @@ inputs:
|
|
|
56
56
|
Ask low-confidence idle sessions directly whether they're done. Off by
|
|
57
57
|
default — it spends a turn in someone else's conversation.
|
|
58
58
|
default: false
|
|
59
|
+
recurring:
|
|
60
|
+
type: boolean
|
|
61
|
+
description: >-
|
|
62
|
+
Set by scheduled runs (the hourly routine). A recurring run closes
|
|
63
|
+
sessions nobody asked about, so a keepAlive session must also have been
|
|
64
|
+
idle `keepAliveAskAfterMinutes`. Default false = an on-demand run, which
|
|
65
|
+
applies no such delay.
|
|
66
|
+
default: false
|
|
67
|
+
keepAliveAskAfterMinutes:
|
|
68
|
+
type: number
|
|
69
|
+
description: >-
|
|
70
|
+
With `askSessions`, for `recurring` runs only: a keepAlive session whose
|
|
71
|
+
worktree is merged or clean (nothing uncommitted, nothing ahead of base)
|
|
72
|
+
is asked only once idle this many minutes; 0 disables the recurring
|
|
73
|
+
keepAlive ask. On-demand runs ignore it. keepAlive only re-lights a
|
|
74
|
+
session after a daemon restart; it does not stop a declared DONE from
|
|
75
|
+
closing it.
|
|
76
|
+
default: 1440
|
|
59
77
|
callerSessionId:
|
|
60
78
|
type: string
|
|
61
79
|
description: The calling session's id — never a candidate.
|
|
@@ -243,7 +261,7 @@ steps:
|
|
|
243
261
|
- id: relabelQueue
|
|
244
262
|
kind: transform
|
|
245
263
|
name: Terminal sessions missing an outcome (relabel candidates)
|
|
246
|
-
description: Entry-based — buildRelabelQueue. PR numbers come from the session row.
|
|
264
|
+
description: Entry-based — buildRelabelQueue. PR numbers (and their owner/repo) come from the session row.
|
|
247
265
|
|
|
248
266
|
- id: relabelEvidenceQueue
|
|
249
267
|
kind: transform
|
|
@@ -266,7 +284,7 @@ steps:
|
|
|
266
284
|
- id: relabelFinal
|
|
267
285
|
kind: transform
|
|
268
286
|
name: Relabel proposals with their evidence
|
|
269
|
-
description: Entry-based — applyRelabelEvidence
|
|
287
|
+
description: Entry-based — applyRelabelEvidence. `done` only when the session's own PR is merged and its last message leaves nothing pending; an open PR or pending work → needs-follow-up; a shared worktree's PR credits nobody.
|
|
270
288
|
|
|
271
289
|
- id: evidence
|
|
272
290
|
kind: map
|
|
@@ -342,7 +360,10 @@ steps:
|
|
|
342
360
|
- id: askQueue
|
|
343
361
|
kind: transform
|
|
344
362
|
name: Low-confidence idle sessions to ask directly
|
|
345
|
-
description:
|
|
363
|
+
description: >-
|
|
364
|
+
Entry-based. Empty unless `askSessions`. A keepAlive session is included
|
|
365
|
+
only with a merged/clean worktree, and on a `recurring` run only once
|
|
366
|
+
idle `keepAliveAskAfterMinutes` (default 1440, 0 = never).
|
|
346
367
|
|
|
347
368
|
- id: ask
|
|
348
369
|
kind: map
|
|
@@ -442,7 +463,8 @@ Every rule below is a pure function in `cron-rules.mjs`, pinned by
|
|
|
442
463
|
`session-steward-cron-rules.test.ts`:
|
|
443
464
|
|
|
444
465
|
- **Loop (1).** `tool_calls_list` per busy session: the same argv verbatim ≥3
|
|
445
|
-
in 10 min, distinct/total < 0.2, or the same file read ≥4
|
|
466
|
+
in 10 min, distinct/total < 0.2, or the same file read ≥4 (a recursive `rg`/`grep` over a
|
|
467
|
+
directory, or its pattern, is not a file read) → `looping`, a
|
|
446
468
|
sub-case of `active`. The proposed action is an **interrupt nudge**, never a
|
|
447
469
|
close. Useful loops (watch, test/type-check re-runs, `git status`, `gh pr`
|
|
448
470
|
polling) are excluded.
|
|
@@ -488,8 +510,16 @@ as observed, never nudged.
|
|
|
488
510
|
- `session_wrapup_apply` re-classifies each id right before acting and always
|
|
489
511
|
refuses `keep`-class ids; this workflow never feeds it one.
|
|
490
512
|
- Rules only ever close `close`/`stuck` ids; a `keepAlive` session is never
|
|
491
|
-
in those classes, so only a confident judge verdict (with
|
|
492
|
-
close it — as FIX-9A allows.
|
|
513
|
+
in those classes, so only a confident judge verdict or a declared DONE (with
|
|
514
|
+
`judgedBy`) can close it — as FIX-9A allows. `keepAlive` means "re-light
|
|
515
|
+
after a daemon restart", not "never close": with `askSessions`, a keepAlive
|
|
516
|
+
session whose worktree is merged or clean (no uncommitted change, nothing
|
|
517
|
+
ahead of base; unknown worktree is not clean) is asked like any other. Closing
|
|
518
|
+
is an active act, so an on-demand run (`recurring: false`, e.g. `agentproto
|
|
519
|
+
steward --ask-sessions`) applies no idle delay to it beyond `idleMinutes`; only
|
|
520
|
+
a `recurring` run waits `keepAliveAskAfterMinutes` (default 1440 = 24 h, 0
|
|
521
|
+
disables), so nobody wakes up to find a session closed that they meant to
|
|
522
|
+
resume. A steward close is a deliberate end, so the sentinel never revives it.
|
|
493
523
|
- A malformed judge reply is `active` with confidence 0 — never acted on.
|
|
494
524
|
- The caller's own session (`callerSessionId`) is dropped from every list.
|
|
495
525
|
- `blocked` / `needs-input` only FLAG a session; it keeps running.
|
|
@@ -64,6 +64,7 @@ export function isUsefulLoopCommand(signature) {
|
|
|
64
64
|
}
|
|
65
65
|
|
|
66
66
|
const READ_TOOLS = new Set(["read", "cat", "rg", "grep", "sed", "head", "tail", "less", "view", "readfile"])
|
|
67
|
+
const FILE_EXT = /\.(md|ts|tsx|js|mjs|cjs|json|txt|py|go|rs|sql|yml|yaml|toml|log|sh)$/
|
|
67
68
|
const READ_VERB_WORD = /^["']?(cat|rg|grep|sed|head|tail|less|view)["']?$/
|
|
68
69
|
|
|
69
70
|
/**
|
|
@@ -83,15 +84,29 @@ export function readTargetOf(record) {
|
|
|
83
84
|
const segments = command !== undefined && !readTool ? command.split(/&&|\|\||\||;/) : [undefined]
|
|
84
85
|
for (const seg of segments) {
|
|
85
86
|
let tokens = seg !== undefined ? seg.trim().split(/\s+/) : command !== undefined ? command.split(/\s+/) : args
|
|
87
|
+
let searchVerb = tool === "rg" || tool === "grep"
|
|
86
88
|
if (seg !== undefined) {
|
|
87
89
|
const first = tokens.findIndex((t) => !t.includes("=") || t.startsWith("-"))
|
|
88
90
|
if (first < 0 || !READ_VERB_WORD.test(tokens[first])) continue
|
|
91
|
+
searchVerb = /^["']?(rg|grep)["']?$/.test(tokens[first])
|
|
89
92
|
tokens = tokens.slice(first + 1)
|
|
90
93
|
}
|
|
94
|
+
// `rg`/`grep` take a pattern first, then search roots: the pattern is not
|
|
95
|
+
// a file, and a root without an extension is a directory (a recursive
|
|
96
|
+
// search of one tree is not "the same file re-read").
|
|
97
|
+
let patternPending = searchVerb
|
|
91
98
|
for (const raw of tokens) {
|
|
92
99
|
const t = raw.replace(/^['"]|['"]$/g, "")
|
|
93
100
|
if (!t || t.startsWith("-") || t.includes("=")) continue
|
|
94
|
-
if (
|
|
101
|
+
if (patternPending) {
|
|
102
|
+
patternPending = false
|
|
103
|
+
continue
|
|
104
|
+
}
|
|
105
|
+
if (searchVerb) {
|
|
106
|
+
if (FILE_EXT.test(t)) return t
|
|
107
|
+
continue
|
|
108
|
+
}
|
|
109
|
+
if (t.includes("/") || FILE_EXT.test(t)) return t
|
|
95
110
|
}
|
|
96
111
|
if (seg === undefined) return args[0] ?? null
|
|
97
112
|
}
|
|
@@ -209,7 +224,8 @@ const minutesSince = (ts, nowMs) => {
|
|
|
209
224
|
|
|
210
225
|
/**
|
|
211
226
|
* A session that never actually ran: explicit 0 in AND 0 out. `undefined`
|
|
212
|
-
* tokens (not reported) is NOT "never ran" — only a hard 0/0
|
|
227
|
+
* tokens (not reported) is NOT "never ran" — only a hard 0/0 with no cost,
|
|
228
|
+
* context use or completed turn is. A session
|
|
213
229
|
* that is busy, still starting/provisioning, or has a prompt queued for its
|
|
214
230
|
* first turn is merely young, not stuck. With `opts.nowMs`, the session must
|
|
215
231
|
* also be at least `opts.idleMinutes` old (`startedAt`) AND idle
|
|
@@ -217,6 +233,9 @@ const minutesSince = (ts, nowMs) => {
|
|
|
217
233
|
*/
|
|
218
234
|
export function isNeverRan(session, opts = {}) {
|
|
219
235
|
if (!(session?.tokensIn === 0 && session?.tokensOut === 0)) return false
|
|
236
|
+
// Some adapters (opencode) report 0/0 tokens for a session that ran several
|
|
237
|
+
// turns; cost, context use or a completed turn prove it ran.
|
|
238
|
+
if (session.costUsd > 0 || session.contextUsed > 0 || session.lastTurnReason === "completed") return false
|
|
220
239
|
if (session.busy === true || session.status === "starting" || session.provisioning) return false
|
|
221
240
|
if (Array.isArray(session.pendingPrompts) && session.pendingPrompts.length > 0) return false
|
|
222
241
|
if (typeof opts.nowMs === "number") {
|
|
@@ -256,7 +275,7 @@ export function isCommitOrPrCall(record) {
|
|
|
256
275
|
|
|
257
276
|
/**
|
|
258
277
|
* The done fast-path: the last tool call is `message_parent(kind:done)` AND
|
|
259
|
-
* the window shows a commit/PR (or
|
|
278
|
+
* the window shows a commit/PR (or a MERGED worktree PR or the derived
|
|
260
279
|
* outcome already proves one). That is `done` without spending a judge.
|
|
261
280
|
* Anything short of both halves returns `{ done: false }`.
|
|
262
281
|
*/
|
|
@@ -267,9 +286,9 @@ export function detectFastPathDone(input = {}) {
|
|
|
267
286
|
if (!doneMessage) return { done: false, reason: "no message_parent(kind:done)" }
|
|
268
287
|
const prState = input.worktree?.pr?.state
|
|
269
288
|
const merged = prState === "merged" || prState === "MERGED"
|
|
270
|
-
const opened = prState === "open" || prState === "OPEN"
|
|
271
289
|
const outcomePrs = Array.isArray(input.outcome?.pullRequests) ? input.outcome.pullRequests.length : 0
|
|
272
|
-
|
|
290
|
+
// An OPEN worktree PR is work awaiting review, not proof of completion.
|
|
291
|
+
const hasCommitOrPr = calls.some(isCommitOrPrCall) || merged || outcomePrs > 0
|
|
273
292
|
if (!hasCommitOrPr) return { done: false, reason: "message_parent(kind:done) but no commit/PR" }
|
|
274
293
|
return { done: true, reason: merged || outcomePrs > 0 ? "message_parent(kind:done) + merged/opened PR" : "message_parent(kind:done) + commit" }
|
|
275
294
|
}
|
|
@@ -296,82 +315,194 @@ export function prNumbersOf(session) {
|
|
|
296
315
|
|
|
297
316
|
/** Relabel verdict for "no positive evidence either way". */
|
|
298
317
|
export const UNKNOWN_VERDICT = "unknown"
|
|
318
|
+
/** Relabel verdict for "a PR/outcome exists but work remains" (report only —
|
|
319
|
+
* the steward never closes on it, and a terminal session has nothing to flag). */
|
|
320
|
+
export const FOLLOW_UP_VERDICT = "needs-follow-up"
|
|
299
321
|
|
|
300
|
-
const fmtPrs = nums => nums.map(n => `#${n}`).join(", ")
|
|
301
322
|
const isMergedState = state => state === "merged" || state === "MERGED"
|
|
323
|
+
const isOpenState = state => state === "open" || state === "OPEN"
|
|
324
|
+
|
|
325
|
+
/** `owner/repo` out of a GitHub PR URL; undefined when it names no repo. */
|
|
326
|
+
export function repoOfPrUrl(url) {
|
|
327
|
+
const m = /github\.com\/([A-Za-z0-9_.-]+\/[A-Za-z0-9_.-]+)\/pull\/\d+/.exec(String(url ?? ""))
|
|
328
|
+
return m ? m[1] : undefined
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
/** `owner/repo#N` when the repo is known, else `#N` — a bare number is
|
|
332
|
+
* ambiguous across repos (G3: `#504` was agentik-studio's, not agentproto's). */
|
|
333
|
+
export const fmtPr = (n, repo) => (repo ? `${repo}#${n}` : `#${n}`)
|
|
334
|
+
const fmtPrs = (nums, repos) => nums.map(n => fmtPr(n, repos?.[n])).join(", ")
|
|
335
|
+
|
|
336
|
+
/** owner/repo per PR number from the session's `outcome.artifacts` PR refs
|
|
337
|
+
* (`…github.com/OWNER/REPO/pull/N`) — the only list-row field that carries
|
|
338
|
+
* a repo. Plain object (journal-safe). */
|
|
339
|
+
export function prReposOf(session) {
|
|
340
|
+
const repos = {}
|
|
341
|
+
for (const a of Array.isArray(session?.outcome?.artifacts) ? session.outcome.artifacts : []) {
|
|
342
|
+
if (a?.type !== "pr") continue
|
|
343
|
+
const m = /\/pull\/(\d+)/.exec(String(a.ref ?? ""))
|
|
344
|
+
const repo = repoOfPrUrl(a.ref)
|
|
345
|
+
if (m && repo) repos[Number(m[1])] = repo
|
|
346
|
+
}
|
|
347
|
+
return repos
|
|
348
|
+
}
|
|
302
349
|
|
|
303
350
|
/**
|
|
304
351
|
* A terminal session that still carries no derived outcome and no wrapup
|
|
305
352
|
* flag is a relabel CANDIDATE — visible instead of invisible, as the log
|
|
306
|
-
* asks.
|
|
307
|
-
*
|
|
308
|
-
*
|
|
309
|
-
*
|
|
310
|
-
*
|
|
353
|
+
* asks. A PR is never proof the session finished, so the proposal is
|
|
354
|
+
* deliberately cautious: `done` only when the session's OWN record shows its
|
|
355
|
+
* PR merged (a worktree PR it never recorded is shared with sibling sessions
|
|
356
|
+
* and credits nobody); otherwise `unknown`. {@link refineRelabel} sharpens
|
|
357
|
+
* it with evidence and runs the remaining-work check.
|
|
311
358
|
*/
|
|
312
359
|
export function terminalRelabelCandidate(session) {
|
|
313
360
|
if (!TERMINAL_STATUSES.has(String(session?.status ?? ""))) return { candidate: false, reason: "not terminal" }
|
|
314
361
|
if (session?.outcome?.verdict) return { candidate: false, reason: "outcome already recorded" }
|
|
315
362
|
if (session?.wrapupFlag) return { candidate: false, reason: "already flagged" }
|
|
316
363
|
const prs = prNumbersOf(session)
|
|
364
|
+
const repos = prReposOf(session)
|
|
317
365
|
const wt = session?.worktree?.pr
|
|
318
|
-
|
|
319
|
-
|
|
320
|
-
// with it, and a PR the session did not record itself is labelled as the
|
|
321
|
-
// worktree's.
|
|
322
|
-
const ownMerged = Number.isInteger(wt?.number) && prs.includes(wt.number)
|
|
323
|
-
if (isMergedState(wt?.state) && (ownMerged || !session?.lastTurnErroredAt)) {
|
|
324
|
-
if (Number.isInteger(wt.number)) {
|
|
325
|
-
return { candidate: true, proposedVerdict: "done", reason: `PR ${fmtPrs([wt.number])} merged${prs.includes(wt.number) ? "" : " (worktree)"}`, prs }
|
|
326
|
-
}
|
|
327
|
-
return { candidate: true, proposedVerdict: "done", reason: prs.length > 0 ? `PR ${fmtPrs(prs)} merged` : "PR merged", prs }
|
|
366
|
+
if (isMergedState(wt?.state) && Number.isInteger(wt?.number) && prs.includes(wt.number)) {
|
|
367
|
+
return { candidate: true, proposedVerdict: "done", reason: `PR ${fmtPr(wt.number, repos[wt.number])} merged`, prs, repos }
|
|
328
368
|
}
|
|
329
369
|
if (prs.length > 0) {
|
|
330
|
-
return { candidate: true, proposedVerdict:
|
|
370
|
+
return { candidate: true, proposedVerdict: UNKNOWN_VERDICT, reason: `PR${prs.length > 1 ? "s" : ""} ${fmtPrs(prs, repos)} recorded, state unknown`, prs, repos }
|
|
331
371
|
}
|
|
332
|
-
return { candidate: true, proposedVerdict: UNKNOWN_VERDICT, reason: "terminal, no PR recorded — outcome unknown", prs }
|
|
372
|
+
return { candidate: true, proposedVerdict: UNKNOWN_VERDICT, reason: "terminal, no PR recorded — outcome unknown", prs, repos }
|
|
333
373
|
}
|
|
334
374
|
|
|
335
375
|
/**
|
|
336
|
-
* Sharpen one relabel proposal with its `session_evidence` answer
|
|
337
|
-
* PR
|
|
338
|
-
*
|
|
339
|
-
*
|
|
376
|
+
* Sharpen one relabel proposal with its `session_evidence` answer.
|
|
377
|
+
* - the worktree PR is merged AND the session recorded it → `done`;
|
|
378
|
+
* - merged but NOT recorded by this session (a shared worktree) → `unknown`;
|
|
379
|
+
* - open and the session's own → `needs-follow-up` (awaiting review/merge);
|
|
380
|
+
* open and not its own → `unknown`;
|
|
381
|
+
* - no PR: `abandoned` only on positive evidence (errored / never ran).
|
|
382
|
+
* Whatever lands on `done` then goes through the remaining-work check on the
|
|
383
|
+
* last assistant turn: a question or announced next step → `needs-follow-up`.
|
|
340
384
|
*/
|
|
341
385
|
export function refineRelabel(item, evidence) {
|
|
342
386
|
if (!evidence) return item
|
|
343
387
|
const wt = evidence.worktree?.pr
|
|
344
|
-
const state = wt?.state ??
|
|
388
|
+
const state = wt?.state ?? null
|
|
345
389
|
const known = Array.isArray(item?.prs) ? item.prs : []
|
|
346
|
-
|
|
347
|
-
|
|
348
|
-
|
|
349
|
-
|
|
350
|
-
const
|
|
351
|
-
const
|
|
352
|
-
|
|
353
|
-
if (
|
|
354
|
-
|
|
355
|
-
|
|
356
|
-
|
|
357
|
-
|
|
358
|
-
|
|
359
|
-
|
|
360
|
-
|
|
361
|
-
if (
|
|
362
|
-
|
|
363
|
-
|
|
364
|
-
}
|
|
365
|
-
const opened = evidence.pullRequests?.opened ?? 0
|
|
366
|
-
if (opened > 0) return { ...item, proposedVerdict: "done", reason: `${opened} PR${opened > 1 ? "s" : ""} opened` }
|
|
367
|
-
// No PR: `abandoned` needs positive evidence the session did not finish.
|
|
368
|
-
if (item?.proposedVerdict === UNKNOWN_VERDICT) {
|
|
390
|
+
const n = Number.isInteger(wt?.number) ? wt.number : undefined
|
|
391
|
+
const repos = { ...(item?.repos ?? {}) }
|
|
392
|
+
const wtRepo = repoOfPrUrl(wt?.url)
|
|
393
|
+
if (n !== undefined && wtRepo) repos[n] = wtRepo
|
|
394
|
+
const label = n !== undefined ? fmtPr(n, repos[n]) : "PR"
|
|
395
|
+
const own = n !== undefined && known.includes(n)
|
|
396
|
+
let next = item
|
|
397
|
+
if (isMergedState(state) && n !== undefined) {
|
|
398
|
+
next = own
|
|
399
|
+
? { ...item, proposedVerdict: "done", reason: `PR ${label} merged` }
|
|
400
|
+
: { ...item, proposedVerdict: UNKNOWN_VERDICT, reason: `worktree PR ${label} merged, not recorded by this session (shared worktree)` }
|
|
401
|
+
} else if (isOpenState(state) && n !== undefined) {
|
|
402
|
+
next = own
|
|
403
|
+
? { ...item, proposedVerdict: FOLLOW_UP_VERDICT, reason: `PR ${label} open — awaiting review/merge` }
|
|
404
|
+
: { ...item, proposedVerdict: UNKNOWN_VERDICT, reason: `worktree PR ${label} open, not recorded by this session` }
|
|
405
|
+
} else if ((evidence.pullRequests?.opened ?? 0) > 0) {
|
|
406
|
+
const opened = evidence.pullRequests.opened
|
|
407
|
+
next = { ...item, proposedVerdict: UNKNOWN_VERDICT, reason: `${opened} PR${opened > 1 ? "s" : ""} opened, state unknown` }
|
|
408
|
+
} else if (item?.proposedVerdict === UNKNOWN_VERDICT) {
|
|
369
409
|
if (typeof evidence.lastTurnError === "string" && evidence.lastTurnError.trim()) {
|
|
370
|
-
|
|
410
|
+
next = { ...item, proposedVerdict: "abandoned", reason: `no PR, last turn errored: ${evidence.lastTurnError.trim().slice(0, 80)}` }
|
|
411
|
+
} else if (evidence.turnsCompleted === 0) {
|
|
412
|
+
next = { ...item, proposedVerdict: "abandoned", reason: "no PR, no turn ever completed" }
|
|
413
|
+
}
|
|
414
|
+
}
|
|
415
|
+
if (Object.keys(repos).length > 0) next = { ...next, repos }
|
|
416
|
+
if (next.proposedVerdict === "done") {
|
|
417
|
+
const turns = Array.isArray(evidence.turns) ? evidence.turns : []
|
|
418
|
+
const lastAssistant = [...turns].reverse().find(t => t?.role === "assistant")
|
|
419
|
+
const pending = remainingWork({ lastAssistantText: lastAssistant?.text })
|
|
420
|
+
if (pending.length > 0) {
|
|
421
|
+
return { ...next, proposedVerdict: FOLLOW_UP_VERDICT, reason: `${next.reason}; remaining work: ${pending.join(", ")}` }
|
|
371
422
|
}
|
|
372
|
-
if (evidence.turnsCompleted === 0) return { ...item, proposedVerdict: "abandoned", reason: "no PR, no turn ever completed" }
|
|
373
423
|
}
|
|
374
|
-
return
|
|
424
|
+
return next
|
|
425
|
+
}
|
|
426
|
+
|
|
427
|
+
// ── remaining-work detection (G11) ───────────────────────────────────────
|
|
428
|
+
//
|
|
429
|
+
// A PR, a merged worktree or an ended parent says the work MOVED ON, never
|
|
430
|
+
// that THIS session finished: a child whose last message asks something or
|
|
431
|
+
// announces a next step has an orphaned obligation once its parent is gone.
|
|
432
|
+
// Closing it silently drops that. These checks therefore turn a would-be
|
|
433
|
+
// close / `done` into a FLAG; they are conservative on purpose (a false
|
|
434
|
+
// positive costs one flag, a false negative drops a question).
|
|
435
|
+
|
|
436
|
+
/** A `?` that ends a sentence, a quote or a bracket — not a URL query (`?a=1`). */
|
|
437
|
+
const QUESTION_MARK = /\?(?=\s|$|["'»”’)\]*_`])/
|
|
438
|
+
/** How far back from the end of the message a `?` still counts as "the
|
|
439
|
+
* question this session left open". */
|
|
440
|
+
const QUESTION_WINDOW_CHARS = 240
|
|
441
|
+
|
|
442
|
+
/** Deferred-work phrasing, EN + FR. `kind` names the signal in the flag note. */
|
|
443
|
+
const PENDING_PATTERNS = [
|
|
444
|
+
{ kind: "open question", re: /\b(open\s+question|unanswered|question\s+ouverte|sans\s+réponse|j'?ai\s+posé\s+la\s+question|the\s+question\s+i\s+(?:put|sent|asked))\b/i },
|
|
445
|
+
{ kind: "asks the user", re: /\b(should\s+i|should\s+we|shall\s+i|shall\s+we|do\s+you\s+want|would\s+you\s+like|dois-je|faut-il|je\s+corrige|voulez-vous|veux-tu)\b/i },
|
|
446
|
+
{ kind: "waiting", re: /\b(waiting\s+(?:for|on)|i'?ll\s+report|i\s+will\s+report|en\s+attente|j'?attends|dans\s+l'?attente)\b/i },
|
|
447
|
+
{ kind: "next step", re: /\b(next\s+steps?|to\s+do\s+next|follow[- ]?ups?|todo|à\s+faire|reste\s+à|prochaines?\s+étapes?|il\s+faudrait|avant\s+le\s+prochain|(?:i|we)\s+(?:still\s+)?(?:need|have)\s+to)\b|\breste\s*:/i },
|
|
448
|
+
{ kind: "recommendation", re: /\b(recommandation|je\s+recommande|i\s+recommend|i'?d\s+recommend|recommendation)\b/i },
|
|
449
|
+
]
|
|
450
|
+
/** "nothing left to do", "rien à faire", "no follow-up" — a negation right
|
|
451
|
+
* before a marker makes it a clean sign-off, not a pending item. */
|
|
452
|
+
const NEGATION_BEFORE = /\b(nothing|no|none|rien|aucune?|n'?ai\s+rien|ne\s+reste\s+rien|no\s+further|no\s+more)\b[^.!?]{0,30}$/i
|
|
453
|
+
|
|
454
|
+
/** The last sentence announces an action rather than reporting a result
|
|
455
|
+
* ("Now drive it with the built CLI.") — the transcript stops mid-work.
|
|
456
|
+
* A sentence that also states an outcome ("Now it works.") does not count. */
|
|
457
|
+
const ANNOUNCES_NEXT = /^(?:now|next|then|ensuite|maintenant|puis|let\s+me(?!\s+know)|let'?s|i'?ll|i\s+will|je\s+vais|on\s+va)\b/i
|
|
458
|
+
const STATES_OUTCOME = /\b(is|are|was|were|works?|passes?|passed|green|done|merged|fixed|complete[d]?|ready|ok|terminé|fini|fonctionne)\b/i
|
|
459
|
+
|
|
460
|
+
function announcesNextAction(flat) {
|
|
461
|
+
const sentences = flat.split(/(?<=[.!?])\s+/).filter(Boolean)
|
|
462
|
+
const last = sentences.at(-1)?.trim()
|
|
463
|
+
return !!last && last.length <= 160 && ANNOUNCES_NEXT.test(last) && !STATES_OUTCOME.test(last)
|
|
464
|
+
}
|
|
465
|
+
|
|
466
|
+
/**
|
|
467
|
+
* Why a session's LAST assistant message still owes something, or `null`
|
|
468
|
+
* when it reads as a clean final report. Reasons: `question to the user`,
|
|
469
|
+
* `announced next action`, `open question`, `asks the user`, `waiting`, `next step`,
|
|
470
|
+
* `recommendation`.
|
|
471
|
+
* The tail the plan carries is ~600 chars, whitespace-flattened.
|
|
472
|
+
*/
|
|
473
|
+
export function pendingWorkReason(text) {
|
|
474
|
+
if (typeof text !== "string" || text.trim().length === 0) return null
|
|
475
|
+
const flat = text.replace(/\s+/g, " ").trim()
|
|
476
|
+
const recent = flat.slice(-QUESTION_WINDOW_CHARS)
|
|
477
|
+
if (QUESTION_MARK.test(recent)) return "question to the user"
|
|
478
|
+
if (announcesNextAction(flat)) return "announced next action"
|
|
479
|
+
for (const { kind, re } of PENDING_PATTERNS) {
|
|
480
|
+
const m = re.exec(flat)
|
|
481
|
+
if (!m) continue
|
|
482
|
+
if (NEGATION_BEFORE.test(flat.slice(Math.max(0, m.index - 40), m.index))) continue
|
|
483
|
+
return kind
|
|
484
|
+
}
|
|
485
|
+
return null
|
|
486
|
+
}
|
|
487
|
+
|
|
488
|
+
/** Boolean form of {@link pendingWorkReason}. */
|
|
489
|
+
export const hasPendingWork = text => pendingWorkReason(text) !== null
|
|
490
|
+
|
|
491
|
+
/** PR states that mean "still waiting on someone" (not merged, not closed). */
|
|
492
|
+
const isOpenPrState = state => state === "open" || state === "OPEN"
|
|
493
|
+
|
|
494
|
+
/**
|
|
495
|
+
* The remaining-work check for one session: reasons it must NOT be closed
|
|
496
|
+
* or labelled `done`. Inputs are whatever the caller has — the last
|
|
497
|
+
* assistant text and/or the worktree's PR state; missing inputs add no
|
|
498
|
+
* reason. An empty list means nothing pending was found.
|
|
499
|
+
*/
|
|
500
|
+
export function remainingWork({ lastAssistantText, prState } = {}) {
|
|
501
|
+
const reasons = []
|
|
502
|
+
const pending = pendingWorkReason(lastAssistantText)
|
|
503
|
+
if (pending) reasons.push(pending)
|
|
504
|
+
if (isOpenPrState(prState)) reasons.push("open PR awaiting review/merge")
|
|
505
|
+
return reasons
|
|
375
506
|
}
|
|
376
507
|
|
|
377
508
|
// ── self-exclusion (mission item 7) ──────────────────────────────────────
|
|
@@ -577,15 +708,26 @@ export function saturationHeader(hostLoad) {
|
|
|
577
708
|
* Why there is nothing to do, stated explicitly instead of an empty report:
|
|
578
709
|
* how many live sessions were seen, how many were busy, how many were
|
|
579
710
|
* terminal (and whether any still need a relabel), and how many were
|
|
580
|
-
* excluded
|
|
711
|
+
* excluded — with a per-reason breakdown (`counts.excludedByReason`) so the
|
|
712
|
+
* report says exactly WHY (G6), e.g. `6 excluded (2 same_cron_job_as_caller,
|
|
713
|
+
* 3 pty, 1 archived)`.
|
|
581
714
|
*/
|
|
582
715
|
export function explainZeroCandidates(counts = {}) {
|
|
583
716
|
const n = (v) => (typeof v === "number" ? v : 0)
|
|
717
|
+
const byReason = counts.excludedByReason
|
|
718
|
+
let excluded = `${n(counts.excluded)} excluded`
|
|
719
|
+
const breakdown = []
|
|
720
|
+
if (byReason && typeof byReason === "object") {
|
|
721
|
+
for (const [reason, count] of Object.entries(byReason)) {
|
|
722
|
+
if (count > 0) breakdown.push(`${count} ${reason}`)
|
|
723
|
+
}
|
|
724
|
+
}
|
|
725
|
+
if (breakdown.length > 0) excluded += ` (${breakdown.join(", ")})`
|
|
584
726
|
return (
|
|
585
727
|
`0 candidates: ${n(counts.live)} live (` +
|
|
586
728
|
`${n(counts.busy)} busy, ${n(counts.idle)} idle), ` +
|
|
587
729
|
`${n(counts.terminal)} terminal (${n(counts.terminalRelabel)} need a relabel), ` +
|
|
588
|
-
|
|
730
|
+
excluded
|
|
589
731
|
)
|
|
590
732
|
}
|
|
591
733
|
|
|
@@ -49,6 +49,7 @@ import {
|
|
|
49
49
|
saturationHeader,
|
|
50
50
|
shouldRejudge,
|
|
51
51
|
refineRelabel,
|
|
52
|
+
remainingWork,
|
|
52
53
|
terminalRelabelCandidate,
|
|
53
54
|
verdictMemoryEvent,
|
|
54
55
|
NUDGE_CONTINUE,
|
|
@@ -57,6 +58,11 @@ import {
|
|
|
57
58
|
|
|
58
59
|
const DEFAULT_IDLE_MINUTES = 30
|
|
59
60
|
const DEFAULT_MIN_CONFIDENCE = 0.8
|
|
61
|
+
/** Recurring (scheduled) runs only: a keepAlive session must have been idle
|
|
62
|
+
* this long before it may be asked, so a session its owner meant to pick up
|
|
63
|
+
* tomorrow is not closed overnight. 0 turns the recurring keepAlive ask off.
|
|
64
|
+
* An on-demand run (`recurring: false`) has no such delay. */
|
|
65
|
+
const DEFAULT_KEEPALIVE_ASK_AFTER_MINUTES = 1440
|
|
60
66
|
/** Terminal sessions older than this (hours since they ended) are not listed
|
|
61
67
|
* as relabel candidates. */
|
|
62
68
|
const DEFAULT_RELABEL_WINDOW_HOURS = 24
|
|
@@ -118,6 +124,8 @@ export function resolveSettings(input, modelRoles) {
|
|
|
118
124
|
jevModel: typeof i.jevModel === "string" && i.jevModel.trim() ? i.jevModel.trim() : DEFAULT_JEV_MODEL,
|
|
119
125
|
maxJudged: Math.floor(num(i.maxJudged, DEFAULT_MAX_JUDGED)),
|
|
120
126
|
askSessions: i.askSessions === true,
|
|
127
|
+
recurring: i.recurring === true,
|
|
128
|
+
keepAliveAskAfterMinutes: num(i.keepAliveAskAfterMinutes, DEFAULT_KEEPALIVE_ASK_AFTER_MINUTES, { min: 0 }),
|
|
121
129
|
callerSessionId: typeof i.callerSessionId === "string" && i.callerSessionId ? i.callerSessionId : null,
|
|
122
130
|
callerOrigin: typeof i.callerOrigin === "string" && i.callerOrigin ? i.callerOrigin : null,
|
|
123
131
|
appId: typeof i.appId === "string" ? i.appId.trim() : DEFAULT_APP_ID,
|
|
@@ -132,11 +140,28 @@ function policyOf(settings) {
|
|
|
132
140
|
return { userOrigins: settings?.userOrigins, closableOrigins: settings?.closableOrigins }
|
|
133
141
|
}
|
|
134
142
|
|
|
143
|
+
/** Why a would-be close must not happen: the remaining-work check over a plan
|
|
144
|
+
* entry — its last assistant message (`signals.lastAssistantTail`) and its
|
|
145
|
+
* worktree's PR still open (`signals.worktreePrOpen`). A merged worktree or an
|
|
146
|
+
* ended parent only says the work moved on, never that THIS session finished
|
|
147
|
+
* (G11: 7 children of an ended parent were closed `done` while their last
|
|
148
|
+
* message asked the parent a question that nobody can answer any more). */
|
|
149
|
+
export function remainingWorkOf(entry) {
|
|
150
|
+
return remainingWork({
|
|
151
|
+
lastAssistantText: entry?.signals?.lastAssistantTail,
|
|
152
|
+
prState: entry?.signals?.worktreePrOpen === true ? "open" : undefined,
|
|
153
|
+
})
|
|
154
|
+
}
|
|
155
|
+
|
|
135
156
|
/** `decideAction` over one candidate entry, with the run's policy folded in.
|
|
136
157
|
* Used by both the apply-queue builders and the report, so the action shown
|
|
137
|
-
* and the action executed can never drift.
|
|
138
|
-
|
|
139
|
-
|
|
158
|
+
* and the action executed can never drift. A rule or judged `close` whose
|
|
159
|
+
* session still has remaining work becomes a `flag` (`remaining` lists why):
|
|
160
|
+
* the operator sees the question instead of the session being closed. A
|
|
161
|
+
* `stuck` session never ran, so it has nothing to owe; a session that itself
|
|
162
|
+
* declared `STEWARD: DONE` (`opts.declared`) answered the question. */
|
|
163
|
+
export function decideFor(entry, planClass, verdict, confidence, settings, opts = {}) {
|
|
164
|
+
const d = decideAction({
|
|
140
165
|
session: entry,
|
|
141
166
|
planClass,
|
|
142
167
|
verdict,
|
|
@@ -145,6 +170,19 @@ export function decideFor(entry, planClass, verdict, confidence, settings) {
|
|
|
145
170
|
policy: policyOf(settings),
|
|
146
171
|
minConfidence: settings?.minConfidence,
|
|
147
172
|
})
|
|
173
|
+
if (d.action !== "close" || planClass === "stuck" || opts.declared === true) return d
|
|
174
|
+
const remaining = remainingWorkOf(entry)
|
|
175
|
+
if (remaining.length === 0) return d
|
|
176
|
+
const reason = `flag (remaining work: ${remaining.join(", ")})`
|
|
177
|
+
return { action: "flag", reason: settings?.apply === true ? reason : `${reason} (dry run)`, remaining }
|
|
178
|
+
}
|
|
179
|
+
|
|
180
|
+
/** The flag note for a decision that downgraded a close: the reason plus the
|
|
181
|
+
* session's own last words, so the operator sees the question it left open. */
|
|
182
|
+
function flagNote(entry, d, extra) {
|
|
183
|
+
const tail = typeof entry?.signals?.lastAssistantTail === "string" ? entry.signals.lastAssistantTail.trim().slice(-300) : ""
|
|
184
|
+
const quoted = d.remaining && tail ? ` — last message: «${tail}»` : ""
|
|
185
|
+
return `${d.reason}${extra ? ` — ${extra}` : ""}${quoted}`
|
|
148
186
|
}
|
|
149
187
|
|
|
150
188
|
// ── plan → candidates ────────────────────────────────────────────────────
|
|
@@ -183,7 +221,7 @@ export function buildRuleApplyQueue(candidates, settings) {
|
|
|
183
221
|
queue.push({
|
|
184
222
|
sessionId: e.sessionId,
|
|
185
223
|
verdict: d.action === "close" ? closeVerdict : "needs-input",
|
|
186
|
-
note: d.action === "close" ? closeNote(e) : d
|
|
224
|
+
note: d.action === "close" ? closeNote(e) : flagNote(e, d),
|
|
187
225
|
})
|
|
188
226
|
}
|
|
189
227
|
}
|
|
@@ -427,8 +465,33 @@ export function collectVerdicts(evidenceResult, judgeQueue, jevResult, jevQueue,
|
|
|
427
465
|
|
|
428
466
|
// ── ask (opt-in) ─────────────────────────────────────────────────────────
|
|
429
467
|
|
|
430
|
-
/**
|
|
431
|
-
*
|
|
468
|
+
/** Nothing to lose in the session's worktree: it is merged (the plan's signal
|
|
469
|
+
* or the PR state), or it is a linked worktree with no uncommitted changes
|
|
470
|
+
* and no commits beyond the base. No worktree info at all is unknown, not
|
|
471
|
+
* clean — a session in a plain checkout stays out. */
|
|
472
|
+
export function worktreeHasNothingToLose(evidence) {
|
|
473
|
+
if (evidence?.signals?.worktreeMerged === true) return true
|
|
474
|
+
const wt = evidence?.worktree
|
|
475
|
+
if (!wt) return false
|
|
476
|
+
if (wt.pr?.state === "merged") return true
|
|
477
|
+
return wt.dirty === false && wt.ahead === 0
|
|
478
|
+
}
|
|
479
|
+
|
|
480
|
+
/** keepAlive is the restart-relight flag; closing is a separate, active
|
|
481
|
+
* decision. A keepAlive session whose worktree holds nothing to lose may be
|
|
482
|
+
* asked. An on-demand run (a human asked for it) applies no idle delay beyond
|
|
483
|
+
* the plan's own `idleMinutes`. A `recurring` run closes things nobody asked
|
|
484
|
+
* about, so it also waits `keepAliveAskAfterMinutes` of idleness (0 = never). */
|
|
485
|
+
export function keepAliveAskEligible(evidence, settings) {
|
|
486
|
+
if (evidence?.keepAlive !== true || !worktreeHasNothingToLose(evidence)) return false
|
|
487
|
+
if (settings.recurring !== true) return true
|
|
488
|
+
const after = settings.keepAliveAskAfterMinutes
|
|
489
|
+
return after > 0 && typeof evidence.idleMinutes === "number" && evidence.idleMinutes >= after
|
|
490
|
+
}
|
|
491
|
+
|
|
492
|
+
/** Sessions to ask directly: judged below `minConfidence`, not awaitingInput
|
|
493
|
+
* or busy, and either not keepAlive or a keepAlive passing
|
|
494
|
+
* {@link keepAliveAskEligible} — only when `askSessions`. */
|
|
432
495
|
export function buildAskQueue(verdicts, settings) {
|
|
433
496
|
if (!settings?.askSessions) return []
|
|
434
497
|
return (verdicts ?? [])
|
|
@@ -436,7 +499,7 @@ export function buildAskQueue(verdicts, settings) {
|
|
|
436
499
|
r =>
|
|
437
500
|
r.evidence &&
|
|
438
501
|
r.confidence < settings.minConfidence &&
|
|
439
|
-
r.evidence.keepAlive !== true &&
|
|
502
|
+
(r.evidence.keepAlive !== true || keepAliveAskEligible(r.evidence, settings)) &&
|
|
440
503
|
r.evidence.awaitingInput !== true &&
|
|
441
504
|
r.evidence.busy !== true,
|
|
442
505
|
)
|
|
@@ -496,14 +559,14 @@ export function buildJudgedApplyQueue(finalVerdicts, settings) {
|
|
|
496
559
|
const queue = []
|
|
497
560
|
for (const r of finalVerdicts ?? []) {
|
|
498
561
|
if (r.malformed) continue
|
|
499
|
-
const d = decideFor(r.entry, "judge", r.verdict, r.confidence, settings)
|
|
562
|
+
const d = decideFor(r.entry, "judge", r.verdict, r.confidence, settings, { declared: r.source === "declared" })
|
|
500
563
|
if (d.action === "skip") continue
|
|
501
564
|
const isFlagVerdict = r.verdict === "blocked" || r.verdict === "needs-input"
|
|
502
565
|
queue.push({
|
|
503
566
|
sessionId: r.entry.sessionId,
|
|
504
567
|
verdict: d.action === "close" ? r.verdict : isFlagVerdict ? r.verdict : "needs-input",
|
|
505
568
|
judgedBy: r.source === "declared" ? `steward-ask:${r.entry.sessionId}` : r.judgedBy ?? r.judgeSessionId ?? "steward-judge",
|
|
506
|
-
note: d.action === "close" ? r.reason :
|
|
569
|
+
note: d.action === "close" ? r.reason : flagNote(r.entry, d, r.reason),
|
|
507
570
|
})
|
|
508
571
|
}
|
|
509
572
|
return queue
|
|
@@ -582,8 +645,10 @@ export function buildReport(b) {
|
|
|
582
645
|
`| ${cls} | ${cell(e.label ?? e.sessionId)} | ${cell(originCell(e, s))} | ${e.idleMinutes ?? "?"} min | ${fmtMB(e.rssBytes)} | ` +
|
|
583
646
|
`${cell(verdict)} | ${conf === undefined ? "—" : conf.toFixed(2)} | ${cell(reason)} | ${cell(action)} |`,
|
|
584
647
|
)
|
|
648
|
+
let withRemainingWork = 0
|
|
585
649
|
for (const e of c.close) {
|
|
586
650
|
const d = decideFor(e, "close", "done", 1, s)
|
|
651
|
+
if (d.remaining) withRemainingWork++
|
|
587
652
|
row("close", e, "done (rules)", undefined, (e.reasons ?? []).join("; "), actionCell(e.sessionId, d, applied))
|
|
588
653
|
}
|
|
589
654
|
for (const e of c.stuck) {
|
|
@@ -591,7 +656,8 @@ export function buildReport(b) {
|
|
|
591
656
|
row("stuck", e, "abandoned (rules)", undefined, (e.reasons ?? []).join("; ") || "stuck starting, never ran", actionCell(e.sessionId, d, applied))
|
|
592
657
|
}
|
|
593
658
|
for (const r of verdicts) {
|
|
594
|
-
const d = decideFor(r.entry, "judge", r.verdict, r.confidence, s)
|
|
659
|
+
const d = decideFor(r.entry, "judge", r.verdict, r.confidence, s, { declared: r.source === "declared" })
|
|
660
|
+
if (d.remaining) withRemainingWork++
|
|
595
661
|
const by = r.source === "declared" ? " (declared)" : r.source === "jev" ? " (jev)" : r.source === "judged" ? " (agent)" : ""
|
|
596
662
|
const reason = r.jevFallback ? `${r.reason} [jev failed: ${r.jevFallback} → agent judge]` : r.reason
|
|
597
663
|
row("judge", r.entry, `${r.verdict}${by}`, r.confidence, reason, actionCell(r.entry.sessionId, d, applied))
|
|
@@ -613,6 +679,7 @@ export function buildReport(b) {
|
|
|
613
679
|
`- candidates: close=${c.close.length} stuck=${c.stuck.length} judge=${verdicts.length}` +
|
|
614
680
|
(c.judgeOverflow?.length ? ` (+${c.judgeOverflow.length} not judged)` : ""),
|
|
615
681
|
)
|
|
682
|
+
if (withRemainingWork > 0) lines.push(`- ${withRemainingWork} would-be close(s) downgraded to a flag: the session still has remaining work (open question, announced next step, open PR)`)
|
|
616
683
|
if (verdicts.length > 0) lines.push(`- verdicts: ${Object.entries(counts).map(([k, n]) => `${k}=${n}`).join(" ")}`)
|
|
617
684
|
const bySource = { jev: 0, agent: 0 }
|
|
618
685
|
let fallbacks = 0
|
|
@@ -749,7 +816,7 @@ export function scanLive(liveSessions, settings, nowMs) {
|
|
|
749
816
|
const endedAt = s.endedAt ?? s.lastActivityAt ?? s.startedAt
|
|
750
817
|
const endedMs = endedAt ? Date.parse(endedAt) : Number.NaN
|
|
751
818
|
if (Number.isFinite(endedMs) && nowMs - endedMs <= relabelWindowMs) {
|
|
752
|
-
terminalRelabel.push({ sessionId: id, origin: s.origin, originClass, label, proposedVerdict: cand.proposedVerdict, reason: cand.reason, prs: cand.prs ?? [], endedAt, endedMs })
|
|
819
|
+
terminalRelabel.push({ sessionId: id, origin: s.origin, originClass, label, proposedVerdict: cand.proposedVerdict, reason: cand.reason, prs: cand.prs ?? [], repos: cand.repos ?? {}, endedAt, endedMs })
|
|
753
820
|
}
|
|
754
821
|
}
|
|
755
822
|
continue
|
|
@@ -779,6 +846,10 @@ export function scanLive(liveSessions, settings, nowMs) {
|
|
|
779
846
|
relabelTotal,
|
|
780
847
|
neverRan: neverRan.length,
|
|
781
848
|
excluded: excluded.length,
|
|
849
|
+
excludedByReason: excluded.reduce((acc, e) => {
|
|
850
|
+
acc[e.reason] = (acc[e.reason] ?? 0) + 1
|
|
851
|
+
return acc
|
|
852
|
+
}, {}),
|
|
782
853
|
}
|
|
783
854
|
return { busy, idle, terminal, terminalRelabel, neverRan, excluded, loopQueue, stallInputs, counts }
|
|
784
855
|
}
|
|
@@ -995,6 +1066,8 @@ export default {
|
|
|
995
1066
|
judgeModel: { type: "string", description: `Model for the agent judge. Default: the \`${ROLE_JUDGE_SESSION}\` model role (repo agentproto.json \`models\` > daemon config \`models\` > built-in).` },
|
|
996
1067
|
maxJudged: { type: "number", description: `Most \`judge\` sessions judged per run, most RAM first. Default ${DEFAULT_MAX_JUDGED}.`, default: DEFAULT_MAX_JUDGED },
|
|
997
1068
|
askSessions: { type: "boolean", description: "Ask low-confidence idle sessions directly whether they're done. Default false — it spends a turn in someone else's conversation.", default: false },
|
|
1069
|
+
recurring: { type: "boolean", description: "Set by scheduled runs (the hourly routine). A recurring run closes sessions nobody asked about, so a keepAlive session must also have been idle `keepAliveAskAfterMinutes`. Default false = an on-demand run, which applies no such delay.", default: false },
|
|
1070
|
+
keepAliveAskAfterMinutes: { type: "number", description: `With \`askSessions\`, for \`recurring\` runs only: a keepAlive session whose worktree is merged or clean (nothing uncommitted, nothing ahead of base) is asked only once idle this many minutes; 0 disables the recurring keepAlive ask. On-demand runs ignore it. keepAlive only re-lights a session after a daemon restart; it does not stop a declared DONE from closing it. Default ${DEFAULT_KEEPALIVE_ASK_AFTER_MINUTES}.`, default: DEFAULT_KEEPALIVE_ASK_AFTER_MINUTES },
|
|
998
1071
|
callerSessionId: { type: "string", description: "The calling session's id — never a candidate. The CLI passes AGENTPROTO_SESSION_ID." },
|
|
999
1072
|
callerOrigin: { type: "string", description: "The calling session's origin (`cron:<jobId>`) — an older run of the SAME cron job is never judged as user work." },
|
|
1000
1073
|
appId: { type: "string", description: `Installed app whose \`app_state\` ledger holds the verdict memory. Default ${DEFAULT_APP_ID}.` },
|
|
@@ -33,10 +33,18 @@ loader path can carry.
|
|
|
33
33
|
judge session is released (killed + archived) when its item settles.
|
|
34
34
|
Verdicts: `done|abandoned|blocked|needs-input|active`. A judge error never
|
|
35
35
|
closes anything.
|
|
36
|
-
5. `ask` (only with `askSessions`) — low-confidence, idle,
|
|
37
|
-
|
|
36
|
+
5. `ask` (only with `askSessions`) — low-confidence, idle, not-awaiting-input
|
|
37
|
+
sessions get ONE prompt asking them to reply
|
|
38
38
|
`STEWARD: DONE …` / `STEWARD: NOT-DONE …`; a ~3 min bounded wait; the
|
|
39
|
-
answer becomes a `declared` verdict.
|
|
39
|
+
answer becomes a `declared` verdict. `keepAlive` only re-lights a session
|
|
40
|
+
after a daemon restart, so it does not exempt one from the ask: a keepAlive
|
|
41
|
+
session whose worktree is merged or clean (nothing uncommitted, nothing
|
|
42
|
+
ahead of base; no worktree info counts as not clean) is asked too. Closing
|
|
43
|
+
is an active act: an on-demand run (`agentproto steward --ask-sessions`)
|
|
44
|
+
adds no idle delay beyond `--idle`. Only a scheduled run (`recurring: true`,
|
|
45
|
+
as the hourly routine sets) waits `keepAliveAskAfterMinutes` (default 1440,
|
|
46
|
+
24 h; 0 turns it off), so a session you meant to resume is not closed
|
|
47
|
+
overnight.
|
|
40
48
|
6. `judgedApply` (only with `apply`) — confident `done`/`abandoned` close
|
|
41
49
|
(resumable, with a recorded outcome); confident `blocked`/`needs-input`
|
|
42
50
|
only flag. Everything else is left alone and reported.
|
|
@@ -59,6 +67,25 @@ the `userOrigins` / `closableOrigins` inputs:
|
|
|
59
67
|
A trailing `*` in either list is a prefix wildcard. The report carries the
|
|
60
68
|
origin column and the retained action in dry run as well as apply.
|
|
61
69
|
|
|
70
|
+
## Remaining work (never close a session that still owes something)
|
|
71
|
+
|
|
72
|
+
A merged worktree, an open or merged PR, or an ended parent says the work
|
|
73
|
+
moved on — not that *this* session finished. Before any rule or judged
|
|
74
|
+
`close`, `decideFor` (`entry.mjs`) runs `remainingWork`
|
|
75
|
+
(`cron-rules.mjs`) over the plan entry's `lastAssistantTail` and
|
|
76
|
+
`worktreePrOpen` signals. A last message that asks a question, proposes a
|
|
77
|
+
next step, announces its next action ("Now drive it…"), recommends something, or waits on someone — or a still-open
|
|
78
|
+
worktree PR — turns the close into a `needs-input` flag whose note quotes the
|
|
79
|
+
session's own last words (`flag (remaining work: …)`). Only a clean final
|
|
80
|
+
report closes. Exceptions: a `stuck` session (it never ran) and a session that
|
|
81
|
+
itself declared `STEWARD: DONE`. Not covered yet: red CI and unanswered review
|
|
82
|
+
comments (the daemon exposes neither).
|
|
83
|
+
|
|
84
|
+
The terminal-session relabel (report only) follows the same principle: `done`
|
|
85
|
+
only when the session's *own* recorded PR is merged and nothing is pending; an
|
|
86
|
+
open PR or pending work is `needs-follow-up`; a worktree PR the session never
|
|
87
|
+
recorded (shared worktree) credits nobody. PRs are reported as `owner/repo#N`.
|
|
88
|
+
|
|
62
89
|
## Running it
|
|
63
90
|
|
|
64
91
|
```bash
|
|
@@ -21,6 +21,9 @@ target:
|
|
|
21
21
|
inputs:
|
|
22
22
|
apply: true
|
|
23
23
|
askSessions: false
|
|
24
|
+
# Scheduled run: if askSessions is ever turned on here, a keepAlive session
|
|
25
|
+
# is asked only after keepAliveAskAfterMinutes idle (default 1440, 0 = never).
|
|
26
|
+
recurring: true
|
|
24
27
|
# Origin policy (the committed default): never close a human's session.
|
|
25
28
|
# `chat-starter`/`vscode` (and any root with no origin and no parent) are
|
|
26
29
|
# FLAG-ONLY; `cron:*` jobs, `gate`, `workflow` and `review` sessions, and executors (a session with
|