@agentproto/apps 0.18.0 → 0.20.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (32) hide show
  1. package/dist/index.d.ts +2 -0
  2. package/dist/index.d.ts.map +1 -1
  3. package/dist/index.mjs +61 -2
  4. package/dist/index.mjs.map +1 -1
  5. package/dist/review-panel/panel.d.ts +1 -1
  6. package/dist/review-panel/panel.d.ts.map +1 -1
  7. package/dist/review-panel/panel.generated.d.ts +1 -1
  8. package/dist/review-panel/panel.generated.d.ts.map +1 -1
  9. package/dist/review-panel/panel.mjs +1 -1
  10. package/dist/review-panel/panel.mjs.map +1 -1
  11. package/dist/review-panel.mjs +1 -1
  12. package/dist/review-panel.mjs.map +1 -1
  13. package/dist/store/index.d.ts +61 -0
  14. package/dist/store/index.d.ts.map +1 -0
  15. package/dist/store/panel.d.ts +53 -0
  16. package/dist/store/panel.d.ts.map +1 -0
  17. package/dist/store/panel.generated.d.ts +13 -0
  18. package/dist/store/panel.generated.d.ts.map +1 -0
  19. package/dist/store/panel.mjs +25 -0
  20. package/dist/store/panel.mjs.map +1 -0
  21. package/dist/store.mjs +71 -0
  22. package/dist/store.mjs.map +1 -0
  23. package/package.json +14 -2
  24. package/session-steward/.agentproto/APP.md +2 -0
  25. package/session-steward/.agentproto/workflows/session-steward/WORKFLOW.md +218 -11
  26. package/session-steward/.agentproto/workflows/session-steward/cron-rules.mjs +526 -0
  27. package/session-steward/.agentproto/workflows/session-steward/entry.mjs +539 -64
  28. package/session-steward/.agentproto/workflows/session-steward/origin-policy.mjs +115 -0
  29. package/session-steward/README.md +19 -2
  30. package/session-steward/routines/session-steward-hourly/ROUTINE.md +24 -1
  31. package/session-steward/scripts/sessions-snapshot.sh +85 -0
  32. package/session-steward/skill/SKILL.md +74 -0
@@ -0,0 +1,115 @@
1
+ // Pure origin policy for the session steward: whether a candidate session may
2
+ // be CLOSED or only FLAGGED is a function of its provenance, so it is a real
3
+ // function here — no I/O, no clock, no daemon state. Imported by entry.mjs
4
+ // (the WORKFLOW.md step graph) and pinned by unit tests. The workflow inputs
5
+ // `userOrigins` / `closableOrigins` configure it; the defaults below are the
6
+ // committed policy: a human is in the loop, so a human-launched session is
7
+ // never closed autonomously.
8
+
9
+ /** Origins the steward must NEVER close — a human launched the session, so a
10
+ * close is always downgraded to a flag. Exact match, or a trailing `*`. */
11
+ export const DEFAULT_USER_ORIGINS = ["chat-starter", "vscode"]
12
+ /** Origins the steward MAY close under the current rules. `cron:*` matches
13
+ * every cron-spawned job (`origin: "cron:<jobId>"`); `gate` matches
14
+ * supervision-gate sessions. */
15
+ export const DEFAULT_CLOSABLE_ORIGINS = ["cron:*", "gate"]
16
+
17
+ /** The reason shown (and recorded) when a would-be close is bounded by origin. */
18
+ export const USER_ORIGIN_REASON = "flag (origine utilisateur)"
19
+
20
+ const isNonEmptyString = (v) => typeof v === "string" && v.trim().length > 0
21
+
22
+ /** One origin-list entry vs a session's origin. A trailing `*` is a prefix
23
+ * wildcard (`cron:*` matches `cron:kill-idle-sessions`); otherwise exact. */
24
+ export function matchesOrigin(origin, pattern) {
25
+ if (!isNonEmptyString(origin) || !isNonEmptyString(pattern)) return false
26
+ const p = pattern.trim()
27
+ if (p.endsWith("*")) return origin.startsWith(p.slice(0, -1))
28
+ return origin === p
29
+ }
30
+
31
+ /** Fold raw (possibly absent/empty/malformed) input lists into a usable
32
+ * policy. An empty list falls back to its default — the conservative
33
+ * direction: a session still has to clear an origin bound to be closed. */
34
+ export function resolveOriginPolicy(policy) {
35
+ const p = policy ?? {}
36
+ const list = (value, fallback) => {
37
+ const kept = Array.isArray(value) ? value.filter(isNonEmptyString).map((s) => s.trim()) : []
38
+ return kept.length > 0 ? kept : fallback
39
+ }
40
+ return {
41
+ userOrigins: list(p.userOrigins, DEFAULT_USER_ORIGINS),
42
+ closableOrigins: list(p.closableOrigins, DEFAULT_CLOSABLE_ORIGINS),
43
+ }
44
+ }
45
+
46
+ /** The origin class of a candidate:
47
+ * - `"user"` — human-launched: a `userOrigins` match, OR a root with no
48
+ * origin and no parent. FLAG ONLY, never close.
49
+ * - `"closable"` — a `closableOrigins` match (cron:*, gate) or an executor
50
+ * (has a `parentSessionId`). Close allowed under the current rules.
51
+ * `userOrigins` wins over both `closableOrigins` and the executor rule, so a
52
+ * `vscode` executor is still user-origin. An unrecognized root origin is
53
+ * treated as `"user"` (conservative). */
54
+ export function classifyOrigin(session, policy) {
55
+ const { userOrigins, closableOrigins } = resolveOriginPolicy(policy)
56
+ const origin = session?.origin
57
+ if (isNonEmptyString(origin) && userOrigins.some((o) => matchesOrigin(origin, o))) return "user"
58
+ if (isNonEmptyString(origin) && closableOrigins.some((o) => matchesOrigin(origin, o))) return "closable"
59
+ if (isNonEmptyString(session?.parentSessionId)) return "closable"
60
+ return "user"
61
+ }
62
+
63
+ const CLOSE_VERDICTS = new Set(["done", "abandoned"])
64
+ const FLAG_VERDICTS = new Set(["blocked", "needs-input"])
65
+
66
+ /**
67
+ * The single decision the steward acts on, pure over its arguments:
68
+ * `{ action: "close" | "flag" | "skip", reason }`. `reason` is the label the
69
+ * report shows and states the retained action even in a dry run (with a
70
+ * `(dry run)` marker), so `action` never depends on `apply`; the queue
71
+ * builders separately gate execution on `apply`.
72
+ *
73
+ * - rule-certain `close`/`stuck` → `close`;
74
+ * - a judge `done`/`abandoned` at or above `minConfidence` → `close`;
75
+ * - a judge `blocked`/`needs-input` at or above `minConfidence` → `flag`;
76
+ * - anything else (`active`, below threshold, unknown verdict/class) → `skip`;
77
+ * - a user-origin candidate is ALWAYS downgraded to `flag` — even a
78
+ * rule-certain close or a confident `done` — with {@link USER_ORIGIN_REASON}.
79
+ */
80
+ export function decideAction({ session, planClass, verdict, confidence, apply, policy, minConfidence } = {}) {
81
+ const originClass = classifyOrigin(session, policy)
82
+ const dry = apply !== true
83
+ const skipped = (reason) => ({ action: "skip", reason: dry ? `${reason} (dry run)` : reason })
84
+
85
+ let action
86
+ let reason
87
+ if (planClass === "close") {
88
+ action = "close"
89
+ reason = "close (règle certaine)"
90
+ } else if (planClass === "stuck") {
91
+ action = "close"
92
+ reason = "close (stuck, jamais démarrée)"
93
+ } else if (planClass === "judge") {
94
+ if (typeof confidence === "number" && typeof minConfidence === "number" && confidence < minConfidence) {
95
+ return skipped(`skip (confiance ${confidence} < ${minConfidence})`)
96
+ }
97
+ if (CLOSE_VERDICTS.has(verdict)) {
98
+ action = "close"
99
+ reason = `close (verdict ${verdict} confiant)`
100
+ } else if (FLAG_VERDICTS.has(verdict)) {
101
+ action = "flag"
102
+ reason = `flag (${verdict})`
103
+ } else {
104
+ return skipped(`skip (verdict ${verdict ?? "inconnu"})`)
105
+ }
106
+ } else {
107
+ return skipped(`skip (classe ${planClass ?? "inconnue"})`)
108
+ }
109
+
110
+ if (originClass === "user" && action === "close") {
111
+ action = "flag"
112
+ reason = USER_ORIGIN_REASON
113
+ }
114
+ return { action, reason: dry ? `${reason} (dry run)` : reason }
115
+ }
@@ -14,7 +14,8 @@ loader path can carry.
14
14
  1. `plan` — `session_wrapup_plan { idleMinutes }` (dry run).
15
15
  2. `autoApply` (only with `apply`) — `close` ids →
16
16
  `session_wrapup_apply { verdict: "done", note: "steward-rules: …" }`,
17
- `stuck` ids → `{ verdict: "abandoned" }`.
17
+ `stuck` ids → `{ verdict: "abandoned" }` — unless the session is
18
+ user-origin, which is flagged instead (see Origin policy).
18
19
  3. `evidence` — per `judge` candidate (at most `maxJudged`, most RAM first),
19
20
  the read-only `session_evidence` tool: label, cwd, idle, keepAlive, RAM,
20
21
  the plan's signals, the last ~10 turns (~3 KB), and worktree
@@ -39,9 +40,25 @@ loader path can carry.
39
40
  6. `judgedApply` (only with `apply`) — confident `done`/`abandoned` close
40
41
  (resumable, with a recorded outcome); confident `blocked`/`needs-input`
41
42
  only flag. Everything else is left alone and reported.
42
- 7. `report` — markdown table (class, session, idle, RAM, verdict,
43
+ 7. `report` — markdown table (class, session, origin, idle, RAM, verdict,
43
44
  confidence, reason, action) plus RAM freed / still held.
44
45
 
46
+ ## Origin policy (never close a human's session)
47
+
48
+ Every candidate's `origin`/`parentSessionId` runs through the pure
49
+ `decideAction` (`workflows/session-steward/origin-policy.mjs`), configured by
50
+ the `userOrigins` / `closableOrigins` inputs:
51
+
52
+ - **Flag only, never close:** `chat-starter`, `vscode`, and any root with no
53
+ `origin` and no `parentSessionId` (a human launched it). A would-be close —
54
+ even a rule-certain `close`/`stuck`, even a confident `done` — is recorded
55
+ as a `needs-input` flag with reason `flag (origine utilisateur)`.
56
+ - **Close allowed:** `cron:*`, `gate`, and executors (a session with a
57
+ `parentSessionId`).
58
+
59
+ A trailing `*` in either list is a prefix wildcard. The report carries the
60
+ origin column and the retained action in dry run as well as apply.
61
+
45
62
  ## Running it
46
63
 
47
64
  ```bash
@@ -21,6 +21,12 @@ target:
21
21
  inputs:
22
22
  apply: true
23
23
  askSessions: false
24
+ # Origin policy (the committed default): never close a human's session.
25
+ # `chat-starter`/`vscode` (and any root with no origin and no parent) are
26
+ # FLAG-ONLY; `cron:*` jobs, `gate` sessions, and executors (a session with
27
+ # a parentSessionId) stay closeable. A trailing `*` is a prefix wildcard.
28
+ userOrigins: ["chat-starter", "vscode"]
29
+ closableOrigins: ["cron:*", "gate"]
24
30
  retry:
25
31
  max_attempts: 1
26
32
  backoff: fixed
@@ -42,7 +48,9 @@ Runs every hour on the hour (UTC), firing the `session-steward` workflow
42
48
 
43
49
  1. Plans with `session_wrapup_plan` (idle ≥ 30 min by default).
44
50
  2. Closes `close`-class sessions as `done` and `stuck`-class ones as
45
- `abandoned` — resumable, with a recorded outcome.
51
+ `abandoned` — resumable, with a recorded outcome — **unless the session is
52
+ user-origin** (`chat-starter`, `vscode`, or a root with no origin and no
53
+ parent), which is flagged instead.
46
54
  3. Judges up to 15 `judge`-class sessions, most RAM first, and closes
47
55
  (`done`/`abandoned`) or flags (`blocked`/`needs-input`) only verdicts at
48
56
  confidence ≥ 0.8. Everything else is left alone and reported.
@@ -51,6 +59,21 @@ Runs every hour on the hour (UTC), firing the `session-steward` workflow
51
59
  awaiting input are never touched (`session_wrapup_apply` re-checks every id
52
60
  right before acting).
53
61
 
62
+ ## Origin policy
63
+
64
+ The steward bounds every action by the candidate's `origin` (pure
65
+ `decideAction`, `workflows/session-steward/origin-policy.mjs`):
66
+
67
+ - **Flag only, never close:** `chat-starter`, `vscode`, and any root with no
68
+ `origin` and no `parentSessionId` (a human launched it). A would-be close —
69
+ even a rule-certain `close`/`stuck`, even a confident `done` — becomes a
70
+ `needs-input` flag with reason `flag (origine utilisateur)`.
71
+ - **Close allowed:** `cron:*` (any cron job), `gate`, and executors (a
72
+ session with a `parentSessionId`).
73
+
74
+ Both lists are workflow inputs (`userOrigins`, `closableOrigins`); the values
75
+ above are the committed default. A trailing `*` is a prefix wildcard.
76
+
54
77
  ## Enabling
55
78
 
56
79
  1. Run it by hand first and read the reports:
@@ -0,0 +1,85 @@
1
+ #!/usr/bin/env bash
2
+ # Session snapshot: the 50 most recently active daemon agent sessions, written
3
+ # to a markdown file. Data comes from the daemon — never hand-edited.
4
+ #
5
+ # Usage: ./scripts/sessions-snapshot.sh [out.md] [count]
6
+ # default out: data/sessions-latest.md, default count: 50
7
+
8
+ set -euo pipefail
9
+ here="$(cd "$(dirname "$0")" && pwd)"
10
+ app_root="$(dirname "$here")"
11
+ out="${1:-$app_root/data/sessions-latest.md}"
12
+ count="${2:-50}"
13
+ mkdir -p "$(dirname "$out")"
14
+
15
+ # GET /sessions returns { sessions: [...] } with full descriptors and ignores
16
+ # ?limit, so sorting + truncation happen here.
17
+ tmp="$(mktemp -d)"
18
+ trap 'rm -rf "$tmp"' EXIT
19
+ curl -sf http://127.0.0.1:18790/sessions -o "$tmp/sessions.json"
20
+ curl -sf "http://127.0.0.1:18790/workflows?limit=200" -o "$tmp/workflows.json" || echo '{}' > "$tmp/workflows.json"
21
+
22
+ node -e '
23
+ const fs = require("fs");
24
+ const [out, count, dir] = process.argv.slice(1);
25
+ const raw = JSON.parse(fs.readFileSync(dir + "/sessions.json", "utf8"));
26
+ const wf = JSON.parse(fs.readFileSync(dir + "/workflows.json", "utf8"));
27
+ const runs = (Array.isArray(wf) ? wf : wf.items || wf.runs || [])
28
+ .filter((r) => r.workflowId === "session-steward")
29
+ .sort((a, b) => String(b.startedAt).localeCompare(String(a.startedAt)));
30
+ const lastRun = runs[0];
31
+ const ts = (s) => Date.parse(s.lastActivityAt || s.lastOutputAt || s.startedAt || 0) || 0;
32
+ const sessions = raw.sessions
33
+ .filter((s) => s.kind === "agent-cli")
34
+ .sort((a, b) => ts(b) - ts(a))
35
+ .slice(0, Number(count));
36
+ const cell = (v) => String(v ?? "").replace(/\|/g, "\\|").replace(/\s+/g, " ").trim();
37
+ const idle = (s) => {
38
+ const sec = s.secondsSinceLastActivity ?? (Date.now() - ts(s)) / 1000;
39
+ if (sec < 3600) return Math.round(sec / 60) + "m";
40
+ if (sec < 86400) return Math.round(sec / 3600) + "h";
41
+ return Math.round(sec / 86400) + "d";
42
+ };
43
+ const status = (s) =>
44
+ s.status + (s.busy ? " (busy)" : "") + (s.awaitingInput ? " (awaiting input)" : "");
45
+ const flag = (s) =>
46
+ s.wrapupFlag ? s.wrapupFlag.verdict + (s.wrapupFlag.note ? ": " + s.wrapupFlag.note : "") : "";
47
+ const outcome = (s) => {
48
+ const o = s.outcome;
49
+ if (!o) return "";
50
+ const t = o.termination;
51
+ return [o.status, t && t.status + (t.reason ? " (" + t.reason + ")" : "")].filter(Boolean).join(" / ");
52
+ };
53
+ const rows = sessions.map((s) =>
54
+ "| " +
55
+ [
56
+ s.id,
57
+ cell(s.label || s.title),
58
+ s.adapterSlug || "-",
59
+ cell(s.model || s.activeModel || "-"),
60
+ status(s),
61
+ idle(s),
62
+ s.parentSessionId || "",
63
+ cell(flag(s)),
64
+ cell(outcome(s)),
65
+ cell((s.activitySummary?.text || "").slice(0, 100)),
66
+ ].join(" | ") +
67
+ " |",
68
+ );
69
+ const flagged = sessions.filter((s) => s.wrapupFlag).length;
70
+ const md = [
71
+ "# Session snapshot — " + new Date().toISOString(),
72
+ "",
73
+ "Last " + rows.length + " agent sessions by activity (of " + raw.sessions.length +
74
+ " daemon sessions). Steward-flagged: " + flagged + ".",
75
+ "Last steward run: " + (lastRun ? lastRun.runId + " (" + lastRun.status + ", " + lastRun.startedAt + ")" : "none found"),
76
+ "Generated by scripts/sessions-snapshot.sh — do not hand-edit.",
77
+ "",
78
+ "| id | label | adapter | model | status | idle | parent | steward flag | outcome | last summary |",
79
+ "|---|---|---|---|---|---|---|---|---|---|",
80
+ ...rows,
81
+ "",
82
+ ].join("\n");
83
+ fs.writeFileSync(out, md);
84
+ console.log("wrote", out, "(" + rows.length + " sessions, " + flagged + " flagged)");
85
+ ' "$out" "$count" "$tmp"
@@ -0,0 +1,74 @@
1
+ ---
2
+ name: session-steward
3
+ description: Wrap up idle agentproto agent sessions — classify them, close the safe ones, judge the ambiguous ones. Use when the user says "clean up idle sessions", "run the steward", "close finished sessions", or asks which sessions are still needed. Dry run by default; closing (--apply) mutates daemon state.
4
+ ---
5
+
6
+ # Session Steward
7
+
8
+ The steward wraps up idle agent sessions. It classifies them with
9
+ `session_wrapup_plan`, closes the rule-certain ones, and sends the ambiguous
10
+ ones through a cheap judge (Jev classifier, fallback: a one-shot LLM agent).
11
+ Dry run by default — nothing is closed unless `--apply`.
12
+
13
+ ## Run it
14
+
15
+ Prefer the CLI (it installs/updates the app and runs the workflow):
16
+
17
+ ```bash
18
+ agentproto steward --wait
19
+ ```
20
+
21
+ Variants:
22
+
23
+ ```bash
24
+ agentproto steward --wait # dry run: plan + verdicts + report
25
+ agentproto steward --apply --wait # close/flag confident verdicts
26
+ agentproto steward --apply --wait --idle 60 # stricter idle threshold (minutes)
27
+ agentproto steward --apply --wait --min-confidence 0.9
28
+ agentproto steward --apply --wait --judge agent # force the LLM judge lane
29
+ agentproto steward --ask-sessions --wait # ask low-confidence sessions directly
30
+ ```
31
+
32
+ `--wait` blocks until the report is ready (a run takes ~1-3 min depending on
33
+ candidate count). Without it, poll with `workflow_status` on the returned
34
+ runId.
35
+
36
+ ## Read the report
37
+
38
+ The final step output is a markdown table: class (close/stuck/judge),
39
+ session label + id, idle time, RAM, verdict, confidence, reason, action.
40
+ Key reading rules:
41
+
42
+ - `close` rows are safe to auto-close; `judge` rows are what the judge saw.
43
+ - Verdicts: `done|abandoned|blocked|needs-input|active`. Only confident
44
+ `done`/`abandoned` close (sessions stay resumable; transcripts preserved).
45
+ `blocked`/`needs-input` only get flagged, never closed.
46
+ - A judge error never closes anything. If many verdicts come back
47
+ `active` confidence 0, the judge lane failed (check Jev key config
48
+ `jev.apiKey` in ~/.agentproto/config.json, env `JEV_API_KEY` fallback).
49
+ - RAM freed / still held is summarized at the end.
50
+
51
+ ## Session snapshot (browsing aid)
52
+
53
+ `scripts/sessions-snapshot.sh` dumps the 50 most recently active daemon agent
54
+ sessions to a markdown table — id, label, adapter, model, status, idle time,
55
+ parent, steward flag, outcome, last summary, plus the last steward run. It
56
+ reads the daemon's HTTP API (`GET /sessions`, `GET /workflows`) and writes a
57
+ generated file; it never mutates daemon state, so it is safe to run any time.
58
+
59
+ ```bash
60
+ ./scripts/sessions-snapshot.sh [out.md] [count]
61
+ # defaults: data/sessions-latest.md, 50
62
+ ```
63
+
64
+ The output lands under `data/` (git-ignored — it is a regenerated snapshot,
65
+ never committed). Use it to eyeball what the steward is about to classify
66
+ before running with `--apply`.
67
+
68
+ ## Rules
69
+
70
+ - NEVER run with `--apply` without explicit user go-ahead. Dry run first,
71
+ show the report, then apply.
72
+ - Never judge the calling session — the CLI already excludes it.
73
+ - If a session is mid-turn, busy, or has background tasks, the plan skips
74
+ it; re-running later is fine and idempotent.