@agentproto/apps 0.20.1 → 0.20.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -64,7 +64,7 @@ export function isUsefulLoopCommand(signature) {
64
64
  }
65
65
 
66
66
  const READ_TOOLS = new Set(["read", "cat", "rg", "grep", "sed", "head", "tail", "less", "view", "readfile"])
67
- const READ_COMMAND = /\b(cat|rg|grep|sed|head|tail|less|view)\b/
67
+ const READ_VERB_WORD = /^["']?(cat|rg|grep|sed|head|tail|less|view)["']?$/
68
68
 
69
69
  /**
70
70
  * The file a read-like call targeted, or `null`. Best-effort: an in-agent
@@ -76,15 +76,26 @@ export function readTargetOf(record) {
76
76
  const tool = String(record?.tool ?? "").toLowerCase()
77
77
  const command = str(record?.command)
78
78
  const args = Array.isArray(record?.args) ? record.args.map(String) : []
79
- const isRead = READ_TOOLS.has(tool) || (command !== undefined && READ_COMMAND.test(command))
80
- if (!isRead) return null
81
- const tokens = command !== undefined ? command.split(/\s+/) : args
82
- for (const raw of tokens) {
83
- const t = raw.replace(/^['"]|['"]$/g, "")
84
- if (!t || t.startsWith("-") || t.includes("=")) continue
85
- if (t.includes("/") || /\.(md|ts|tsx|js|mjs|cjs|json|txt|py|go|rs|sql|yml|yaml|toml)$/.test(t)) return t
79
+ const readTool = READ_TOOLS.has(tool)
80
+ if (!readTool && command === undefined) return null
81
+ // A shell line is several commands: only a segment that STARTS with a read
82
+ // verb reads a file. `cd <dir> &&`, `git log | head` and `--grep=…` do not.
83
+ const segments = command !== undefined && !readTool ? command.split(/&&|\|\||\||;/) : [undefined]
84
+ for (const seg of segments) {
85
+ let tokens = seg !== undefined ? seg.trim().split(/\s+/) : command !== undefined ? command.split(/\s+/) : args
86
+ if (seg !== undefined) {
87
+ const first = tokens.findIndex((t) => !t.includes("=") || t.startsWith("-"))
88
+ if (first < 0 || !READ_VERB_WORD.test(tokens[first])) continue
89
+ tokens = tokens.slice(first + 1)
90
+ }
91
+ for (const raw of tokens) {
92
+ const t = raw.replace(/^['"]|['"]$/g, "")
93
+ if (!t || t.startsWith("-") || t.includes("=")) continue
94
+ if (t.includes("/") || /\.(md|ts|tsx|js|mjs|cjs|json|txt|py|go|rs|sql|yml|yaml|toml)$/.test(t)) return t
95
+ }
96
+ if (seg === undefined) return args[0] ?? null
86
97
  }
87
- return args[0] ?? null
98
+ return null
88
99
  }
89
100
 
90
101
  function topEntry(counts) {
@@ -115,7 +126,11 @@ export function detectLoop(records, opts = {}) {
115
126
  const ts = Date.parse(r?.ts)
116
127
  return Number.isFinite(ts) && ts <= nowMs + 1000 && nowMs - ts <= windowMs
117
128
  })
118
- const calls = inWindow.filter((r) => !isUsefulLoopCommand(callSignature(r)))
129
+ // A record with no command and no args (an in-agent `read`/`edit` call) has
130
+ // nothing but its tool name to compare, so three of them in ten minutes is
131
+ // normal work, not a loop — leave it out of every repetition signal.
132
+ const informative = inWindow.filter((r) => str(r?.command) !== undefined || (Array.isArray(r?.args) && r.args.length > 0))
133
+ const calls = informative.filter((r) => !isUsefulLoopCommand(callSignature(r)))
119
134
  const counts = new Map()
120
135
  const reads = new Map()
121
136
  for (const r of calls) {
@@ -146,7 +161,8 @@ export function detectLoop(records, opts = {}) {
146
161
  ratio: Math.round(ratio * 100) / 100,
147
162
  maxVerbatim,
148
163
  maxReads,
149
- usefulExcluded: inWindow.length - total,
164
+ usefulExcluded: informative.length - total,
165
+ anonymousExcluded: inWindow.length - informative.length,
150
166
  topCommand: top ? top.key : null,
151
167
  topCommandCount: top ? top.count : 0,
152
168
  },
@@ -262,22 +278,100 @@ export function detectFastPathDone(input = {}) {
262
278
 
263
279
  const TERMINAL_STATUSES = new Set(["killed", "exited", "error", "stopped", "completed", "failed"])
264
280
 
281
+ /** PR numbers a `session_list` row records: `openedPrs[].number` plus the
282
+ * `outcome.artifacts` `pr` entries (`#N` title or a `…/pull/N` ref). Sorted,
283
+ * de-duplicated. */
284
+ export function prNumbersOf(session) {
285
+ const nums = new Set()
286
+ for (const pr of Array.isArray(session?.openedPrs) ? session.openedPrs : []) {
287
+ if (Number.isInteger(pr?.number)) nums.add(pr.number)
288
+ }
289
+ for (const a of Array.isArray(session?.outcome?.artifacts) ? session.outcome.artifacts : []) {
290
+ if (a?.type !== "pr") continue
291
+ const m = /#(\d+)$/.exec(String(a.title ?? "")) ?? /\/pull\/(\d+)/.exec(String(a.ref ?? ""))
292
+ if (m) nums.add(Number(m[1]))
293
+ }
294
+ return [...nums].sort((a, b) => a - b)
295
+ }
296
+
297
+ /** Relabel verdict for "no positive evidence either way". */
298
+ export const UNKNOWN_VERDICT = "unknown"
299
+
300
+ const fmtPrs = nums => nums.map(n => `#${n}`).join(", ")
301
+ const isMergedState = state => state === "merged" || state === "MERGED"
302
+
265
303
  /**
266
304
  * A terminal session that still carries no derived outcome and no wrapup
267
305
  * flag is a relabel CANDIDATE — visible instead of invisible, as the log
268
- * asks. The proposed verdict is `done` when the worktree/PR proves the work
269
- * merged, else `abandoned` (the ambiguous ones still go to a judge).
306
+ * asks. The proposed verdict is `done` when the session's own record shows a
307
+ * PR (merged, or merely opened — the PR is the hand-off), else `unknown`: a
308
+ * session with no PR is as likely finished as abandoned, and only evidence
309
+ * ({@link refineRelabel}) can say which.
310
+ * `reason` carries the evidence (`PR #1738 merged`, `PRs #1738, #1740 opened`).
270
311
  */
271
312
  export function terminalRelabelCandidate(session) {
272
313
  if (!TERMINAL_STATUSES.has(String(session?.status ?? ""))) return { candidate: false, reason: "not terminal" }
273
314
  if (session?.outcome?.verdict) return { candidate: false, reason: "outcome already recorded" }
274
315
  if (session?.wrapupFlag) return { candidate: false, reason: "already flagged" }
275
- const prState = session?.worktree?.pr?.state
276
- const merged = prState === "merged" || prState === "MERGED"
277
- const outcomePrs = Array.isArray(session?.outcome?.pullRequests) ? session.outcome.pullRequests : []
278
- const mergedPr = merged || outcomePrs.some((p) => p?.state === "MERGED" || p?.state === "merged")
279
- if (mergedPr) return { candidate: true, proposedVerdict: "done", reason: "terminal, PR merged, no outcome recorded" }
280
- return { candidate: true, proposedVerdict: "abandoned", reason: "terminal, no outcome recorded" }
316
+ const prs = prNumbersOf(session)
317
+ const wt = session?.worktree?.pr
318
+ // A worktree can be shared by several sessions, so its merged PR is not
319
+ // proof THIS one finished: a session whose last turn errored is not credited
320
+ // with it, and a PR the session did not record itself is labelled as the
321
+ // worktree's.
322
+ const ownMerged = Number.isInteger(wt?.number) && prs.includes(wt.number)
323
+ if (isMergedState(wt?.state) && (ownMerged || !session?.lastTurnErroredAt)) {
324
+ if (Number.isInteger(wt.number)) {
325
+ return { candidate: true, proposedVerdict: "done", reason: `PR ${fmtPrs([wt.number])} merged${prs.includes(wt.number) ? "" : " (worktree)"}`, prs }
326
+ }
327
+ return { candidate: true, proposedVerdict: "done", reason: prs.length > 0 ? `PR ${fmtPrs(prs)} merged` : "PR merged", prs }
328
+ }
329
+ if (prs.length > 0) {
330
+ return { candidate: true, proposedVerdict: "done", reason: `PR${prs.length > 1 ? "s" : ""} ${fmtPrs(prs)} opened`, prs }
331
+ }
332
+ return { candidate: true, proposedVerdict: UNKNOWN_VERDICT, reason: "terminal, no PR recorded — outcome unknown", prs }
333
+ }
334
+
335
+ /**
336
+ * Sharpen one relabel proposal with its `session_evidence` answer: a merged
337
+ * PR / merged worktree (`pullRequests.merged`, `worktree.pr.state`) →
338
+ * `done` with `PR #N merged`; an open or recorded PR the list row did not
339
+ * carry → `done`. Anything else leaves the proposal untouched.
340
+ */
341
+ export function refineRelabel(item, evidence) {
342
+ if (!evidence) return item
343
+ const wt = evidence.worktree?.pr
344
+ const state = wt?.state ?? evidence.pullRequests?.state ?? null
345
+ const known = Array.isArray(item?.prs) ? item.prs : []
346
+ // `pullRequests.merged` and `worktree.pr` describe the whole worktree, which
347
+ // sibling sessions share: a session that recorded no PR of its own and whose
348
+ // last turn errored is not credited with a sibling's merge.
349
+ const ownPr = known.length > 0 || (evidence.pullRequests?.opened ?? 0) > 0
350
+ const errored = typeof evidence.lastTurnError === "string" && evidence.lastTurnError.trim() !== ""
351
+ const credited = ownPr || !errored
352
+ const merged = credited && (isMergedState(state) || (evidence.pullRequests?.merged ?? 0) > 0)
353
+ if (merged) {
354
+ const n = Number.isInteger(wt?.number) ? wt.number : known.length === 1 ? known[0] : undefined
355
+ const others = known.filter(k => k !== n)
356
+ // A PR number the session did not record itself is the worktree's.
357
+ const shared = n !== undefined && !known.includes(n) ? " (worktree)" : ""
358
+ const reason = (n !== undefined ? `PR #${n} merged${shared}` : "PR merged") + (others.length > 0 && n !== undefined ? `; also opened ${fmtPrs(others)}` : "")
359
+ return { ...item, proposedVerdict: "done", reason }
360
+ }
361
+ if (item?.proposedVerdict === "done") return item
362
+ if (credited && (state === "open" || state === "OPEN")) {
363
+ return { ...item, proposedVerdict: "done", reason: Number.isInteger(wt?.number) ? `PR #${wt.number} open` : "PR open" }
364
+ }
365
+ const opened = evidence.pullRequests?.opened ?? 0
366
+ if (opened > 0) return { ...item, proposedVerdict: "done", reason: `${opened} PR${opened > 1 ? "s" : ""} opened` }
367
+ // No PR: `abandoned` needs positive evidence the session did not finish.
368
+ if (item?.proposedVerdict === UNKNOWN_VERDICT) {
369
+ if (typeof evidence.lastTurnError === "string" && evidence.lastTurnError.trim()) {
370
+ return { ...item, proposedVerdict: "abandoned", reason: `no PR, last turn errored: ${evidence.lastTurnError.trim().slice(0, 80)}` }
371
+ }
372
+ if (evidence.turnsCompleted === 0) return { ...item, proposedVerdict: "abandoned", reason: "no PR, no turn ever completed" }
373
+ }
374
+ return item
281
375
  }
282
376
 
283
377
  // ── self-exclusion (mission item 7) ──────────────────────────────────────
@@ -12,9 +12,10 @@
12
12
  // RE-CLASSIFIES each id itself immediately before acting and refuses
13
13
  // `keep`-class ids outright. On top of that, this workflow:
14
14
  // - mutates no SESSION unless `apply` is true (every session-mutating map
15
- // runs over an empty list otherwise). The one dry-run write is the
16
- // append-only verdict-memory ledger (`app_state_append`) — never a
17
- // session, and disableable with `appId: ""`;
15
+ // runs over an empty list otherwise). A dry run writes nothing at all:
16
+ // the append-only verdict-memory ledger (`app_state_append`) is read but
17
+ // only appended to when `apply` is true, so a dry run's verdicts can
18
+ // never be served as cached verdicts to a later real pass;
18
19
  // - only ever feeds `close`/`stuck` ids to the rules pass — a `keepAlive`
19
20
  // session can only ever be `judge` class, so rules never close it;
20
21
  // - drops the caller's own session from every candidate list;
@@ -47,6 +48,7 @@ import {
47
48
  isSelfExcluded,
48
49
  saturationHeader,
49
50
  shouldRejudge,
51
+ refineRelabel,
50
52
  terminalRelabelCandidate,
51
53
  verdictMemoryEvent,
52
54
  NUDGE_CONTINUE,
@@ -641,7 +643,7 @@ export function buildReport(b) {
641
643
  }
642
644
 
643
645
  // Terminal sessions with no outcome (mission item 5).
644
- const relabel = b.steps.relabelQueue ?? []
646
+ const relabel = b.steps.relabelFinal ?? b.steps.relabelQueue ?? []
645
647
  if (relabel.length > 0) {
646
648
  const total = b.steps.scan?.counts?.relabelTotal ?? relabel.length
647
649
  const byVerdict = new Map()
@@ -649,6 +651,7 @@ export function buildReport(b) {
649
651
  const perLabel = [...byVerdict].map(([v, n]) => `${v}: ${n}`).join(", ")
650
652
  lines.push("", "## Terminal sessions missing an outcome (relabel candidates)")
651
653
  lines.push(`- ${relabel.length} ended in the last ${s.relabelWindowHours}h (${total} without an outcome in all) — ${perLabel}`)
654
+ lines.push(`- evidence: PR numbers come from the session record; worktree/PR state was looked up for the newest ${RELABEL_MAX_LINES} only`)
652
655
  for (const r of relabel.slice(0, RELABEL_MAX_LINES)) lines.push(`- ${r.sessionId} → ${r.proposedVerdict} — ${r.reason}`)
653
656
  const hidden = total - Math.min(relabel.length, RELABEL_MAX_LINES)
654
657
  if (hidden > 0) lines.push(`- … and ${hidden} more (older/omitted)`)
@@ -662,6 +665,9 @@ export function buildReport(b) {
662
665
  } else if (memory instanceof Map && memory.size > 0) {
663
666
  lines.push(`- verdict memory: ${memory.size} session(s) known` + (cachedCount > 0 ? `, ${cachedCount} served from cache` : ""))
664
667
  }
668
+ if (!s.apply && b.steps.memoryApp?.appId) {
669
+ lines.push("- verdict memory was read but not written (dry run)")
670
+ }
665
671
 
666
672
  return lines.join("\n")
667
673
  }
@@ -743,7 +749,7 @@ export function scanLive(liveSessions, settings, nowMs) {
743
749
  const endedAt = s.endedAt ?? s.lastActivityAt ?? s.startedAt
744
750
  const endedMs = endedAt ? Date.parse(endedAt) : Number.NaN
745
751
  if (Number.isFinite(endedMs) && nowMs - endedMs <= relabelWindowMs) {
746
- terminalRelabel.push({ sessionId: id, origin: s.origin, originClass, label, proposedVerdict: cand.proposedVerdict, reason: cand.reason, endedAt, endedMs })
752
+ terminalRelabel.push({ sessionId: id, origin: s.origin, originClass, label, proposedVerdict: cand.proposedVerdict, reason: cand.reason, prs: cand.prs ?? [], endedAt, endedMs })
747
753
  }
748
754
  }
749
755
  continue
@@ -802,6 +808,24 @@ export function mergeNeverRan(candidates, scan) {
802
808
  }
803
809
  }
804
810
 
811
+ /** A rule-certain `close` whose session's last turn errored did not finish —
812
+ * parentEnded / merged-worktree only says the parent moved on. Closing it
813
+ * would record `done` on a failed run, so demote it to the judge list. */
814
+ export function demoteErroredCloses(candidates, scan) {
815
+ const errored = new Set((scan?.idle ?? []).filter(r => r?.lastTurnErroredAt).map(r => r.sessionId))
816
+ const close = candidates?.close ?? []
817
+ const demoted = close.filter(e => errored.has(e.sessionId))
818
+ if (demoted.length === 0) return candidates
819
+ return {
820
+ ...candidates,
821
+ close: close.filter(e => !errored.has(e.sessionId)),
822
+ judge: [
823
+ ...(candidates?.judge ?? []),
824
+ ...demoted.map(e => ({ ...e, class: "judge", reasons: [...(e.reasons ?? []), "last turn errored — not auto-closed"] })),
825
+ ],
826
+ }
827
+ }
828
+
805
829
  /** `tool_calls_list` map item → the loop verdict + stats for one session. */
806
830
  export function analyzeLoopItem(b) {
807
831
  const item = b.item ?? {}
@@ -898,10 +922,35 @@ export function buildRelabelQueue(scan) {
898
922
  return (scan?.terminalRelabel ?? []).map(t => ({ ...t })).sort((a, b) => b.endedMs - a.endedMs)
899
923
  }
900
924
 
901
- /** The `app_state` events to append for this pass's verdicts. The memory is
902
- * written on every pass (it is a ledger, never a session action) so streaks
903
- * accumulate and the cache can engage. */
925
+ /** The sessions that get a per-session `session_evidence` lookup: only the
926
+ * newest {@link RELABEL_MAX_LINES} of the (already windowed) relabel queue —
927
+ * the ones the report lists — never every terminal session. */
928
+ export function buildRelabelEvidenceQueue(relabelQueue) {
929
+ return (relabelQueue ?? []).slice(0, RELABEL_MAX_LINES).map(r => ({ sessionId: r.sessionId }))
930
+ }
931
+
932
+ /** One relabel evidence map item: the `session_evidence` answer, id-checked. */
933
+ function foldRelabelEvidence(b) {
934
+ const raw = b.steps.relabelEvidenceOne
935
+ if (!raw || raw.sessionId !== b.item?.sessionId) {
936
+ throw new Error(`session_evidence answered for '${raw?.sessionId}', expected '${b.item?.sessionId}'`)
937
+ }
938
+ return { sessionId: b.item.sessionId, evidence: raw }
939
+ }
940
+
941
+ /** Relabel proposals sharpened by the evidence lookups (a failed lookup
942
+ * leaves its proposal as the list row had it). Order is preserved. */
943
+ export function applyRelabelEvidence(relabelQueue, evidenceResult) {
944
+ const bySession = new Map(settled(evidenceResult).ok.map(r => [r.value?.sessionId, r.value?.evidence]))
945
+ return (relabelQueue ?? []).map(r => (bySession.has(r.sessionId) ? refineRelabel(r, bySession.get(r.sessionId)) : r))
946
+ }
947
+
948
+ /** The `app_state` events to append for this pass's verdicts. Written only on
949
+ * an `apply` pass (so streaks accumulate and the cache can engage); a dry run
950
+ * reads memory but never writes it, or its verdicts would be reused by the
951
+ * next real pass instead of being re-judged. */
904
952
  export function buildMemoryWriteQueue(finalVerdicts, settings) {
953
+ if (!settings?.apply) return []
905
954
  if (!settings?.appId) return []
906
955
  const out = []
907
956
  for (const r of finalVerdicts ?? []) {
@@ -975,7 +1024,7 @@ export default {
975
1024
  { id: "liveSessions", kind: "tool", tool: "session_list", inputs: { full: true } },
976
1025
  { id: "scan", kind: "transform", compute: b => scanLive(b.steps.liveSessions, b.steps.settings, Date.now()) },
977
1026
  // Never-ran 0/0 sessions are `stuck` immediately, never judged (item 3).
978
- { id: "candidatesPlus", kind: "transform", compute: b => mergeNeverRan(b.steps.candidates, b.steps.scan) },
1027
+ { id: "candidatesPlus", kind: "transform", compute: b => demoteErroredCloses(mergeNeverRan(b.steps.candidates, b.steps.scan), b.steps.scan) },
979
1028
  { id: "ruleApplyQueue", kind: "transform", compute: b => buildRuleApplyQueue(b.steps.candidatesPlus, b.steps.settings) },
980
1029
  {
981
1030
  // Empty unless `apply` — a dry run dispatches no apply call at all.
@@ -1031,6 +1080,19 @@ export default {
1031
1080
  // continue, at most one per session per pass, user origins observed only.
1032
1081
  { id: "proposals", kind: "transform", compute: b => buildProposalsStep(b.steps.scan, b.steps.loopResults, b.steps.settings, Date.now()) },
1033
1082
  { id: "relabelQueue", kind: "transform", compute: b => buildRelabelQueue(b.steps.scan) },
1083
+ { id: "relabelEvidenceQueue", kind: "transform", compute: b => buildRelabelEvidenceQueue(b.steps.relabelQueue) },
1084
+ {
1085
+ id: "relabelEvidence",
1086
+ kind: "map",
1087
+ over: "$steps.relabelEvidenceQueue",
1088
+ parallelism: 4,
1089
+ onError: "collect",
1090
+ steps: [
1091
+ { id: "relabelEvidenceOne", kind: "tool", tool: "session_evidence", inputs: { sessionId: "$item.sessionId" } },
1092
+ { id: "relabelEvidenceFold", kind: "transform", compute: foldRelabelEvidence },
1093
+ ],
1094
+ },
1095
+ { id: "relabelFinal", kind: "transform", compute: b => applyRelabelEvidence(b.steps.relabelQueue, b.steps.relabelEvidence) },
1034
1096
  {
1035
1097
  id: "evidence",
1036
1098
  kind: "map",
@@ -1172,7 +1234,8 @@ export default {
1172
1234
  ],
1173
1235
  },
1174
1236
  // Verdict memory write-back (item 10) — a ledger append, never a session
1175
- // action; best-effort (an uninstalled app just yields no memory).
1237
+ // action; empty unless `apply`; best-effort (an uninstalled app just
1238
+ // yields no memory).
1176
1239
  { id: "memoryWriteQueue", kind: "transform", compute: b => buildMemoryWriteQueue(b.steps.finalVerdicts, { ...b.steps.settings, appId: b.steps.memoryApp?.appId ?? "" }) },
1177
1240
  {
1178
1241
  id: "memoryWrite",
@@ -1194,7 +1257,7 @@ export default {
1194
1257
  autoApply: "$steps.autoApply",
1195
1258
  judgedApply: "$steps.judgedApply",
1196
1259
  proposals: "$steps.proposals",
1197
- relabel: "$steps.relabelQueue",
1260
+ relabel: "$steps.relabelFinal",
1198
1261
  scan: "$steps.scan",
1199
1262
  },
1200
1263
  }