@agentproto/apps 0.15.0 → 0.17.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/index.mjs +1 -1
- package/dist/index.mjs.map +1 -1
- package/dist/review-panel/panel.d.ts +11 -2
- package/dist/review-panel/panel.d.ts.map +1 -1
- package/dist/review-panel/panel.generated.d.ts +1 -1
- package/dist/review-panel/panel.generated.d.ts.map +1 -1
- package/dist/review-panel/panel.mjs +1 -1
- package/dist/review-panel/panel.mjs.map +1 -1
- package/dist/review-panel.mjs +1 -1
- package/dist/review-panel.mjs.map +1 -1
- package/package.json +8 -7
- package/repo-maintenance/.agentproto/agents/repo-maintenance-reviewer/AGENT.md +1 -1
- package/repo-maintenance/.agentproto/workflows/maintain/WORKFLOW.md +24 -6
- package/repo-maintenance/.agentproto/workflows/maintain/entry.mjs +35 -9
- package/repo-maintenance/README.md +7 -2
- package/repo-maintenance/routines/repo-maintenance-daily/ROUTINE.md +2 -2
- package/session-steward/.agentproto/APP.md +20 -0
- package/session-steward/.agentproto/agents/session-steward-judge/AGENT.md +43 -0
- package/session-steward/.agentproto/workflows/session-steward/WORKFLOW.md +277 -0
- package/session-steward/.agentproto/workflows/session-steward/entry.mjs +705 -0
- package/session-steward/README.md +70 -0
- package/session-steward/routines/session-steward-hourly/ROUTINE.md +68 -0
|
@@ -26,12 +26,16 @@ inputs:
|
|
|
26
26
|
default: false
|
|
27
27
|
reviewModelSmall:
|
|
28
28
|
type: string
|
|
29
|
-
description:
|
|
30
|
-
|
|
29
|
+
description: >-
|
|
30
|
+
Model for a review candidate with residualFileCount <= 3. Default: the
|
|
31
|
+
`review.small` model role (repo agentproto.json `models` > daemon config
|
|
32
|
+
`models` > built-in).
|
|
31
33
|
reviewModelLarge:
|
|
32
34
|
type: string
|
|
33
|
-
description:
|
|
34
|
-
|
|
35
|
+
description: >-
|
|
36
|
+
Model for a review candidate with residualFileCount > 3, and the retry
|
|
37
|
+
reviewer. Default: the `review.large` model role (repo agentproto.json
|
|
38
|
+
`models` > daemon config `models` > built-in).
|
|
35
39
|
maxReviews:
|
|
36
40
|
type: number
|
|
37
41
|
description: >-
|
|
@@ -47,6 +51,20 @@ inputs:
|
|
|
47
51
|
the report when set. Omit for no notification.
|
|
48
52
|
outputs: {}
|
|
49
53
|
steps:
|
|
54
|
+
- id: modelRoles
|
|
55
|
+
kind: tool
|
|
56
|
+
name: Resolve the reviewer model roles
|
|
57
|
+
tool: model_roles
|
|
58
|
+
inputs:
|
|
59
|
+
repoRoot: $input.repoRoot
|
|
60
|
+
workspaceSlug: $input.workspaceSlug
|
|
61
|
+
roles:
|
|
62
|
+
- review.small
|
|
63
|
+
- review.large
|
|
64
|
+
inputs:
|
|
65
|
+
review.small: $input.reviewModelSmall
|
|
66
|
+
review.large: $input.reviewModelLarge
|
|
67
|
+
|
|
50
68
|
- id: worktreeGcPlan
|
|
51
69
|
kind: tool
|
|
52
70
|
name: Plan worktree gc
|
|
@@ -96,8 +114,8 @@ steps:
|
|
|
96
114
|
name: Review every unmerged branch
|
|
97
115
|
description: >-
|
|
98
116
|
One reviewer-agent turn per branch name, parallelism 4. Model is
|
|
99
|
-
picked per item by entry.mjs:
|
|
100
|
-
|
|
117
|
+
picked per item by entry.mjs: sonnet when residualFileCount <= 3, else
|
|
118
|
+
opus (the `review.small` / `review.large` model roles). The agent records its verdict via branch_gc_verdict — one call
|
|
101
119
|
PER tip the item carries (an item may hold a local + a remote tip that
|
|
102
120
|
diverged); this step never applies anything. After the turn,
|
|
103
121
|
branch_gc_verdict_get checks the store for EVERY tip; with any
|
|
@@ -4,7 +4,7 @@
|
|
|
4
4
|
// only checks the top-level id/kind sequence, never nested step bodies).
|
|
5
5
|
//
|
|
6
6
|
// This has to be entry-based for ONE reason: the `review` map step's
|
|
7
|
-
// per-candidate `model` selector picks
|
|
7
|
+
// per-candidate `model` selector picks the small vs large reviewer model from `item
|
|
8
8
|
// .residualFileCount` at RUN time. `@agentproto/workflow`'s `defineWorkflow`
|
|
9
9
|
// (the pure-.md / TS-authored path) hard-rejects a non-string `model` on a
|
|
10
10
|
// `kind:"agent"` step — "a run-time selector is only available on the
|
|
@@ -19,8 +19,22 @@
|
|
|
19
19
|
import { tmpdir } from "node:os"
|
|
20
20
|
import { join } from "node:path"
|
|
21
21
|
|
|
22
|
-
|
|
23
|
-
|
|
22
|
+
// The reviewer models are model ROLES, not ids here: the `modelRoles` step
|
|
23
|
+
// (the daemon's `model_roles` tool) resolves `review.small` / `review.large`
|
|
24
|
+
// at run time — explicit input > repo agentproto.json `models` > daemon
|
|
25
|
+
// config `models` > built-in default. The one precedence + default table is
|
|
26
|
+
// packages/runtime/src/model-roles.ts; nothing below hard-codes a model id.
|
|
27
|
+
const ROLE_REVIEW_SMALL = "review.small"
|
|
28
|
+
const ROLE_REVIEW_LARGE = "review.large"
|
|
29
|
+
|
|
30
|
+
/** The model for `role`: the run's explicit input (kept ahead of the tool as a
|
|
31
|
+
* belt-and-braces — the tool folds the same input in as its top layer) or
|
|
32
|
+
* the `modelRoles` step's resolution. Undefined leaves the agent's own
|
|
33
|
+
* AGENT.md `model` in charge. */
|
|
34
|
+
export function reviewModel(b, role) {
|
|
35
|
+
const explicit = role === ROLE_REVIEW_SMALL ? b.input?.reviewModelSmall : b.input?.reviewModelLarge
|
|
36
|
+
return explicit || b.steps?.modelRoles?.models?.[role] || undefined
|
|
37
|
+
}
|
|
24
38
|
/** Reviews per run — 500+ candidates in one run is hours of agent turns;
|
|
25
39
|
* daily runs walk the backlog instead (a reviewed tip is skipped next time). */
|
|
26
40
|
const DEFAULT_MAX_REVIEWS = 40
|
|
@@ -451,8 +465,8 @@ export default {
|
|
|
451
465
|
repoRoot: { type: "string", description: "Absolute path to the git repo. Wins over `workspaceSlug`." },
|
|
452
466
|
workspaceSlug: { type: "string", description: "Workspace slug. The active workspace when both are omitted." },
|
|
453
467
|
applyMerged: { type: "boolean", description: "Execute branch_gc/worktree_gc (reclaim-class only) after review. Default false.", default: false },
|
|
454
|
-
reviewModelSmall: { type: "string", description: `Model for a review candidate with residualFileCount <= 3. Default
|
|
455
|
-
reviewModelLarge: { type: "string", description: `Model for a review candidate with residualFileCount > 3. Default
|
|
468
|
+
reviewModelSmall: { type: "string", description: `Model for a review candidate with residualFileCount <= 3. Default: the \`${ROLE_REVIEW_SMALL}\` model role (repo agentproto.json \`models\` > daemon config \`models\` > built-in).` },
|
|
469
|
+
reviewModelLarge: { type: "string", description: `Model for a review candidate with residualFileCount > 3, and the retry reviewer. Default: the \`${ROLE_REVIEW_LARGE}\` model role (repo agentproto.json \`models\` > daemon config \`models\` > built-in).` },
|
|
456
470
|
maxReviews: { type: "number", description: `Most review candidates to review this run — newest tip first, then the larger residual. The rest are reported as not reviewed this run; tips with a stored verdict are never re-reviewed, so daily runs walk the backlog. Default ${DEFAULT_MAX_REVIEWS}.`, default: DEFAULT_MAX_REVIEWS },
|
|
457
471
|
notify: {
|
|
458
472
|
type: "object",
|
|
@@ -461,6 +475,20 @@ export default {
|
|
|
461
475
|
},
|
|
462
476
|
outputs: {},
|
|
463
477
|
steps: [
|
|
478
|
+
{
|
|
479
|
+
id: "modelRoles",
|
|
480
|
+
kind: "tool",
|
|
481
|
+
tool: "model_roles",
|
|
482
|
+
inputs: {
|
|
483
|
+
repoRoot: "$input.repoRoot",
|
|
484
|
+
workspaceSlug: "$input.workspaceSlug",
|
|
485
|
+
roles: [ROLE_REVIEW_SMALL, ROLE_REVIEW_LARGE],
|
|
486
|
+
inputs: {
|
|
487
|
+
[ROLE_REVIEW_SMALL]: "$input.reviewModelSmall",
|
|
488
|
+
[ROLE_REVIEW_LARGE]: "$input.reviewModelLarge",
|
|
489
|
+
},
|
|
490
|
+
},
|
|
491
|
+
},
|
|
464
492
|
{
|
|
465
493
|
id: "worktreeGcPlan",
|
|
466
494
|
kind: "tool",
|
|
@@ -505,9 +533,7 @@ export default {
|
|
|
505
533
|
agent: { ref: REVIEWER_REF },
|
|
506
534
|
cwd: REVIEWER_CWD,
|
|
507
535
|
prompt: REVIEW_PROMPT,
|
|
508
|
-
model: b => ((b.item?.residualFileCount ?? 0) <= 3
|
|
509
|
-
? (b.input?.reviewModelSmall || DEFAULT_REVIEW_MODEL_SMALL)
|
|
510
|
-
: (b.input?.reviewModelLarge || DEFAULT_REVIEW_MODEL_LARGE)),
|
|
536
|
+
model: b => reviewModel(b, (b.item?.residualFileCount ?? 0) <= 3 ? ROLE_REVIEW_SMALL : ROLE_REVIEW_LARGE),
|
|
511
537
|
},
|
|
512
538
|
// A reviewer can end its turn announcing calls it never made. Check
|
|
513
539
|
// the store for EVERY tip of this item (the per-tip map below); if
|
|
@@ -548,7 +574,7 @@ export default {
|
|
|
548
574
|
REVIEW_PROMPT +
|
|
549
575
|
"\n\nA previous reviewer of this branch ended its turn without recording a verdict. " +
|
|
550
576
|
"Recording the verdict with branch_gc_verdict IS the deliverable — do not end your turn without it.",
|
|
551
|
-
model: b => b
|
|
577
|
+
model: b => reviewModel(b, ROLE_REVIEW_LARGE),
|
|
552
578
|
},
|
|
553
579
|
{
|
|
554
580
|
id: "reviewSettled",
|
|
@@ -18,8 +18,13 @@ The `maintain` workflow, one run:
|
|
|
18
18
|
`maxReviews` (default 40) per run, newest tip first, then the larger
|
|
19
19
|
residual; the rest wait for the next run (one turn per unique tip sha,
|
|
20
20
|
parallelism 4, each reviewer in its own disposable detached worktree
|
|
21
|
-
under the OS tmp dir, never the live checkout):
|
|
22
|
-
candidate's residual is 3 files or fewer,
|
|
21
|
+
under the OS tmp dir, never the live checkout): sonnet when the
|
|
22
|
+
candidate's residual is 3 files or fewer, opus otherwise — by default;
|
|
23
|
+
the models are the `review.small` / `review.large` **model roles** (an
|
|
24
|
+
explicit `reviewModelSmall`/`reviewModelLarge` input wins, then the repo's
|
|
25
|
+
`agentproto.json` `models`, then the daemon config `models`, then the
|
|
26
|
+
built-in default — see `@agentproto/runtime`'s `model-roles.ts`; inspect
|
|
27
|
+
with the `model_roles` tool, change with `config_set models.review.small`). Three spawn
|
|
23
28
|
failures in a row stop the fan-out (the engine's circuit breaker) and
|
|
24
29
|
the report lists every distinct failure reason. Each turn records a verdict via `branch_gc_verdict` —
|
|
25
30
|
recording a verdict never deletes anything. A turn that ends without a
|
|
@@ -44,8 +44,8 @@ with `applyMerged: false`. Every run:
|
|
|
44
44
|
|
|
45
45
|
1. Plans `worktree_gc` and `branch_gc` (dry run — nothing is touched).
|
|
46
46
|
2. Fans the `@agentproto/repo-maintenance-reviewer` agent out over every
|
|
47
|
-
unmerged branch candidate (
|
|
48
|
-
|
|
47
|
+
unmerged branch candidate (sonnet when its residual is 3 files or fewer,
|
|
48
|
+
opus otherwise), one turn per unique tip sha. Each turn records a
|
|
49
49
|
verdict via `branch_gc_verdict` — recording never deletes anything.
|
|
50
50
|
3. Re-plans to confirm every candidate got a verdict, and reports the gaps.
|
|
51
51
|
4. Reports a markdown summary, and notifies (when `notify` is configured)
|
|
@@ -0,0 +1,20 @@
|
|
|
1
|
+
---
|
|
2
|
+
schema: app/v1
|
|
3
|
+
id: '@agentproto/session-steward'
|
|
4
|
+
name: Session Steward
|
|
5
|
+
version: 0.1.0
|
|
6
|
+
description: >-
|
|
7
|
+
Wraps up idle agent sessions: classifies them with session_wrapup_plan,
|
|
8
|
+
closes the rule-certain ones, has a cheap one-shot judge decide the
|
|
9
|
+
ambiguous ones from compact evidence, and closes or flags confident
|
|
10
|
+
verdicts with a recorded outcome. Dry run by default.
|
|
11
|
+
agents:
|
|
12
|
+
- id: '@agentproto/session-steward-judge'
|
|
13
|
+
path: .agentproto/agents/session-steward-judge/AGENT.md
|
|
14
|
+
workflows:
|
|
15
|
+
- id: session-steward
|
|
16
|
+
path: .agentproto/workflows/session-steward/WORKFLOW.md
|
|
17
|
+
---
|
|
18
|
+
|
|
19
|
+
Wraps up idle agent sessions — see README.md for how to install and run it,
|
|
20
|
+
and `routines/` for the hourly scheduled template (ships disabled).
|
|
@@ -0,0 +1,43 @@
|
|
|
1
|
+
---
|
|
2
|
+
schema: agent/v1
|
|
3
|
+
id: '@agentproto/session-steward-judge'
|
|
4
|
+
description: >-
|
|
5
|
+
Decides, from a compact evidence object in its prompt, whether ONE idle
|
|
6
|
+
agent session is done, abandoned, blocked, needs input, or still active.
|
|
7
|
+
One turn, one strict JSON verdict, then the session is released. Spawned
|
|
8
|
+
by the `session-steward` workflow's `judge` map step.
|
|
9
|
+
model: role:judge.session
|
|
10
|
+
boundaries:
|
|
11
|
+
- Answer from the evidence in the prompt alone — never call a tool, never read or write files
|
|
12
|
+
- Reply with exactly one JSON object and nothing else
|
|
13
|
+
- When unsure, answer `active` with a low confidence
|
|
14
|
+
tools:
|
|
15
|
+
- session_evidence
|
|
16
|
+
workflows:
|
|
17
|
+
- ref: session-steward
|
|
18
|
+
---
|
|
19
|
+
|
|
20
|
+
You are the session steward's judge. Each prompt carries the evidence for ONE
|
|
21
|
+
idle AI coding-agent session: its label, cwd, idle time, RAM, the planner's
|
|
22
|
+
signals (last assistant message, pending tool call, parent ended, worktree
|
|
23
|
+
merged), its last few turns, and — for a worktree session — branch, dirty
|
|
24
|
+
counts, ahead/behind and PR state.
|
|
25
|
+
|
|
26
|
+
Decide one verdict:
|
|
27
|
+
|
|
28
|
+
- `done` — the task visibly finished: a PR was opened or merged, a final
|
|
29
|
+
report was given, or the user said thanks/ok with nothing pending.
|
|
30
|
+
- `abandoned` — superseded or a dead end, with nothing worth keeping.
|
|
31
|
+
- `blocked` — waiting on something external (CI, another session, a
|
|
32
|
+
dependency).
|
|
33
|
+
- `needs-input` — waiting on a human answer or decision.
|
|
34
|
+
- `active` — mid-work; keep it.
|
|
35
|
+
|
|
36
|
+
Closing a session that still had work is worse than leaving an idle one open:
|
|
37
|
+
when unsure, answer `active` with a low confidence.
|
|
38
|
+
|
|
39
|
+
Reply with ONLY the JSON object the prompt asks for — `sessionId`, `verdict`,
|
|
40
|
+
`confidence` (0..1), `reason` (one line). No prose, no code fence.
|
|
41
|
+
|
|
42
|
+
The one tool on your gateway (`session_evidence`) is there only because an
|
|
43
|
+
agent that declares no tools gets the full daemon gateway; you do not need it.
|
|
@@ -0,0 +1,277 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: Session Steward
|
|
3
|
+
id: session-steward
|
|
4
|
+
description: >-
|
|
5
|
+
Plan idle-session wrap-up (session_wrapup_plan), close the rule-certain
|
|
6
|
+
`close`/`stuck` sessions, judge the ambiguous `judge` ones with a cheap
|
|
7
|
+
one-shot model over compact evidence, optionally ask a session directly,
|
|
8
|
+
then close or flag the confident verdicts with a recorded outcome — and
|
|
9
|
+
report. Dry run unless `apply` is true. Entry-based (see entry.mjs): every
|
|
10
|
+
decision between the tool calls is a real function (candidate split,
|
|
11
|
+
strict verdict parse, confidence threshold, report).
|
|
12
|
+
version: 0.1.0
|
|
13
|
+
entry: ./entry.mjs
|
|
14
|
+
inputs:
|
|
15
|
+
idleMinutes:
|
|
16
|
+
type: number
|
|
17
|
+
description: Idle threshold in minutes.
|
|
18
|
+
default: 30
|
|
19
|
+
apply:
|
|
20
|
+
type: boolean
|
|
21
|
+
description: Close/flag sessions. False = dry run (plan + verdicts, no mutation).
|
|
22
|
+
default: false
|
|
23
|
+
minConfidence:
|
|
24
|
+
type: number
|
|
25
|
+
description: Judge confidence needed to act on a verdict.
|
|
26
|
+
default: 0.8
|
|
27
|
+
judge:
|
|
28
|
+
type: string
|
|
29
|
+
description: >-
|
|
30
|
+
Judge backend — `auto` (Jev when JEV_API_KEY resolves, else the agent
|
|
31
|
+
judge), `jev`, or `agent`. A Jev failure always falls back to the agent
|
|
32
|
+
judge for that session.
|
|
33
|
+
default: auto
|
|
34
|
+
jevModel:
|
|
35
|
+
type: string
|
|
36
|
+
description: Jev model.
|
|
37
|
+
default: jev-latest
|
|
38
|
+
judgeModel:
|
|
39
|
+
type: string
|
|
40
|
+
description: >-
|
|
41
|
+
Model for the agent judge. Default: the `judge.session` model role
|
|
42
|
+
(repo agentproto.json `models` > daemon config `models` > built-in).
|
|
43
|
+
maxJudged:
|
|
44
|
+
type: number
|
|
45
|
+
description: Most `judge` sessions judged per run, most RAM first.
|
|
46
|
+
default: 15
|
|
47
|
+
askSessions:
|
|
48
|
+
type: boolean
|
|
49
|
+
description: >-
|
|
50
|
+
Ask low-confidence idle sessions directly whether they're done. Off by
|
|
51
|
+
default — it spends a turn in someone else's conversation.
|
|
52
|
+
default: false
|
|
53
|
+
callerSessionId:
|
|
54
|
+
type: string
|
|
55
|
+
description: The calling session's id — never a candidate.
|
|
56
|
+
outputs: {}
|
|
57
|
+
steps:
|
|
58
|
+
- id: modelRoles
|
|
59
|
+
kind: tool
|
|
60
|
+
name: Resolve the judge model role
|
|
61
|
+
tool: model_roles
|
|
62
|
+
inputs:
|
|
63
|
+
roles:
|
|
64
|
+
- judge.session
|
|
65
|
+
inputs:
|
|
66
|
+
judge.session: $input.judgeModel
|
|
67
|
+
- id: settings
|
|
68
|
+
kind: transform
|
|
69
|
+
name: Resolve inputs with their defaults
|
|
70
|
+
description: Entry-based — see entry.mjs's resolveSettings.
|
|
71
|
+
|
|
72
|
+
- id: plan
|
|
73
|
+
kind: tool
|
|
74
|
+
name: Classify idle sessions (dry run)
|
|
75
|
+
tool: session_wrapup_plan
|
|
76
|
+
inputs:
|
|
77
|
+
idleMinutes: $steps.settings.idleMinutes
|
|
78
|
+
|
|
79
|
+
- id: candidates
|
|
80
|
+
kind: transform
|
|
81
|
+
name: Split close / stuck / judge, drop keep and the caller
|
|
82
|
+
description: >-
|
|
83
|
+
Entry-based — splitCandidates. `judge` is ordered most RAM first and
|
|
84
|
+
capped at `maxJudged`.
|
|
85
|
+
|
|
86
|
+
- id: ruleApplyQueue
|
|
87
|
+
kind: transform
|
|
88
|
+
name: Rule verdicts to apply
|
|
89
|
+
description: >-
|
|
90
|
+
Entry-based. Empty unless `apply`: `close` → done, `stuck` → abandoned.
|
|
91
|
+
|
|
92
|
+
- id: autoApply
|
|
93
|
+
kind: map
|
|
94
|
+
name: Close rule-certain sessions
|
|
95
|
+
over: $steps.ruleApplyQueue
|
|
96
|
+
parallelism: 1
|
|
97
|
+
onError: collect
|
|
98
|
+
steps:
|
|
99
|
+
- id: autoApplyOne
|
|
100
|
+
kind: tool
|
|
101
|
+
tool: session_wrapup_apply
|
|
102
|
+
inputs:
|
|
103
|
+
sessionIds: [$item.sessionId]
|
|
104
|
+
verdict: $item.verdict
|
|
105
|
+
note: $item.note
|
|
106
|
+
|
|
107
|
+
- id: evidence
|
|
108
|
+
kind: map
|
|
109
|
+
name: Collect compact evidence per judge candidate
|
|
110
|
+
over: $steps.candidates.judge
|
|
111
|
+
parallelism: 4
|
|
112
|
+
onError: collect
|
|
113
|
+
steps:
|
|
114
|
+
- id: evidenceOne
|
|
115
|
+
kind: tool
|
|
116
|
+
tool: session_evidence
|
|
117
|
+
inputs:
|
|
118
|
+
sessionId: $item.sessionId
|
|
119
|
+
|
|
120
|
+
- id: judgeQueue
|
|
121
|
+
kind: transform
|
|
122
|
+
name: Candidates with evidence
|
|
123
|
+
description: Entry-based.
|
|
124
|
+
|
|
125
|
+
- id: jevQueue
|
|
126
|
+
kind: transform
|
|
127
|
+
name: Candidates for the Jev backend
|
|
128
|
+
description: Entry-based. Empty when `judge` is `agent`.
|
|
129
|
+
|
|
130
|
+
- id: jevJudge
|
|
131
|
+
kind: map
|
|
132
|
+
name: Jev judge per candidate
|
|
133
|
+
description: >-
|
|
134
|
+
One `session_judge_jev` call per candidate — a calibrated `choice` over
|
|
135
|
+
the five verdicts with probabilities, state = the evidence. A missing
|
|
136
|
+
key or any failure is `ok:false`, never an error, and that candidate
|
|
137
|
+
goes to the agent judge.
|
|
138
|
+
over: $steps.jevQueue
|
|
139
|
+
parallelism: 4
|
|
140
|
+
onError: collect
|
|
141
|
+
steps:
|
|
142
|
+
- id: jevOne
|
|
143
|
+
kind: tool
|
|
144
|
+
tool: session_judge_jev
|
|
145
|
+
inputs:
|
|
146
|
+
sessionId: $item.entry.sessionId
|
|
147
|
+
evidence: $item.evidence
|
|
148
|
+
model: $steps.settings.jevModel
|
|
149
|
+
|
|
150
|
+
- id: agentJudgeQueue
|
|
151
|
+
kind: transform
|
|
152
|
+
name: Candidates Jev didn't answer
|
|
153
|
+
description: Entry-based — buildAgentJudgeQueue.
|
|
154
|
+
|
|
155
|
+
- id: judge
|
|
156
|
+
kind: map
|
|
157
|
+
name: One-shot agent judge per remaining candidate
|
|
158
|
+
description: >-
|
|
159
|
+
One turn of `@agentproto/session-steward-judge` on `judgeModel`, evidence
|
|
160
|
+
in the prompt, strict JSON verdict out. A malformed reply is `active`
|
|
161
|
+
with confidence 0. The judge session is released (killed + archived)
|
|
162
|
+
when its item settles.
|
|
163
|
+
over: $steps.agentJudgeQueue
|
|
164
|
+
parallelism: 3
|
|
165
|
+
onError: collect
|
|
166
|
+
steps:
|
|
167
|
+
- id: judgeOne
|
|
168
|
+
kind: agent
|
|
169
|
+
agent:
|
|
170
|
+
ref: "@agentproto/session-steward-judge"
|
|
171
|
+
prompt: $item.judgePrompt
|
|
172
|
+
|
|
173
|
+
- id: verdicts
|
|
174
|
+
kind: transform
|
|
175
|
+
name: One verdict row per judged candidate
|
|
176
|
+
description: Entry-based — collectVerdicts.
|
|
177
|
+
|
|
178
|
+
- id: askQueue
|
|
179
|
+
kind: transform
|
|
180
|
+
name: Low-confidence idle sessions to ask directly
|
|
181
|
+
description: Entry-based. Empty unless `askSessions`.
|
|
182
|
+
|
|
183
|
+
- id: ask
|
|
184
|
+
kind: map
|
|
185
|
+
name: Ask the session itself (opt-in)
|
|
186
|
+
description: >-
|
|
187
|
+
One `agent_prompt` (queue:false), a bounded ~3 min `session_monitor`
|
|
188
|
+
wait, then its newest assistant turn is parsed for `STEWARD: DONE` /
|
|
189
|
+
`STEWARD: NOT-DONE`. See entry.mjs.
|
|
190
|
+
over: $steps.askQueue
|
|
191
|
+
parallelism: 2
|
|
192
|
+
onError: collect
|
|
193
|
+
steps:
|
|
194
|
+
- id: askPrompt
|
|
195
|
+
kind: tool
|
|
196
|
+
tool: agent_prompt
|
|
197
|
+
inputs:
|
|
198
|
+
sessionId: $item.sessionId
|
|
199
|
+
|
|
200
|
+
- id: finalVerdicts
|
|
201
|
+
kind: transform
|
|
202
|
+
name: Merge declared answers over judge verdicts
|
|
203
|
+
description: Entry-based — mergeDeclared.
|
|
204
|
+
|
|
205
|
+
- id: judgedApplyQueue
|
|
206
|
+
kind: transform
|
|
207
|
+
name: Confident verdicts to apply
|
|
208
|
+
description: >-
|
|
209
|
+
Entry-based. Empty unless `apply`: done/abandoned/blocked/needs-input at
|
|
210
|
+
or above `minConfidence`.
|
|
211
|
+
|
|
212
|
+
- id: judgedApply
|
|
213
|
+
kind: map
|
|
214
|
+
name: Close or flag judged sessions
|
|
215
|
+
over: $steps.judgedApplyQueue
|
|
216
|
+
parallelism: 1
|
|
217
|
+
onError: collect
|
|
218
|
+
steps:
|
|
219
|
+
- id: judgedApplyOne
|
|
220
|
+
kind: tool
|
|
221
|
+
tool: session_wrapup_apply
|
|
222
|
+
inputs:
|
|
223
|
+
sessionIds: [$item.sessionId]
|
|
224
|
+
verdict: $item.verdict
|
|
225
|
+
judgedBy: $item.judgedBy
|
|
226
|
+
note: $item.note
|
|
227
|
+
|
|
228
|
+
- id: report
|
|
229
|
+
kind: transform
|
|
230
|
+
name: Build the markdown report
|
|
231
|
+
description: Entry-based — buildReport.
|
|
232
|
+
|
|
233
|
+
result:
|
|
234
|
+
report: $steps.report
|
|
235
|
+
apply: $steps.settings.apply
|
|
236
|
+
candidates: $steps.candidates
|
|
237
|
+
verdicts: $steps.finalVerdicts
|
|
238
|
+
autoApply: $steps.autoApply
|
|
239
|
+
judgedApply: $steps.judgedApply
|
|
240
|
+
---
|
|
241
|
+
|
|
242
|
+
# Session Steward — `session-steward` workflow
|
|
243
|
+
|
|
244
|
+
`session_wrapup_plan` → rules pass over `close`/`stuck` → compact evidence per
|
|
245
|
+
`judge` session → one cheap judge turn each → (opt-in) ask the session itself →
|
|
246
|
+
close or flag confident verdicts through `session_wrapup_apply` → markdown
|
|
247
|
+
report with RAM freed / still held.
|
|
248
|
+
|
|
249
|
+
## Safety
|
|
250
|
+
|
|
251
|
+
- `apply: false` (the default) mutates nothing: every mutating map runs over
|
|
252
|
+
an empty list.
|
|
253
|
+
- `session_wrapup_apply` re-classifies each id right before acting and always
|
|
254
|
+
refuses `keep`-class ids; this workflow never feeds it one.
|
|
255
|
+
- Rules only ever close `close`/`stuck` ids; a `keepAlive` session is never
|
|
256
|
+
in those classes, so only a confident judge verdict (with `judgedBy`) can
|
|
257
|
+
close it — as FIX-9A allows.
|
|
258
|
+
- A malformed judge reply is `active` with confidence 0 — never acted on.
|
|
259
|
+
- The caller's own session (`callerSessionId`) is dropped from every list.
|
|
260
|
+
- `blocked` / `needs-input` only FLAG a session; it keeps running.
|
|
261
|
+
|
|
262
|
+
## The judge
|
|
263
|
+
|
|
264
|
+
Two backends. **Jev** (TypeSafe System One, `session_judge_jev`) is the
|
|
265
|
+
default whenever `JEV_API_KEY` resolves (daemon env, else the host secret
|
|
266
|
+
resolver): one calibrated `choice` over the five verdicts; confidence is the
|
|
267
|
+
chosen verdict's probability, the full probabilities go in the report, and
|
|
268
|
+
`judgedBy` is `jev:<model>`. Any Jev failure (no key under `judge: jev`, a
|
|
269
|
+
non-2xx after retries, a malformed answer) falls back to the agent judge for
|
|
270
|
+
that session and the report says so — a Jev error never closes anything.
|
|
271
|
+
|
|
272
|
+
The **agent judge**:
|
|
273
|
+
|
|
274
|
+
`@agentproto/session-steward-judge` answers from the evidence in its prompt
|
|
275
|
+
and is told not to call tools. Its gateway mount is scoped to the one
|
|
276
|
+
read-only `session_evidence` tool: declaring no `tools:` at all would mount
|
|
277
|
+
the FULL daemon gateway for a claude-code judge (the host's default).
|