@agentproto/apps 0.20.0 → 0.20.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -47,6 +47,7 @@ import {
47
47
  isSelfExcluded,
48
48
  saturationHeader,
49
49
  shouldRejudge,
50
+ refineRelabel,
50
51
  terminalRelabelCandidate,
51
52
  verdictMemoryEvent,
52
53
  NUDGE_CONTINUE,
@@ -55,6 +56,11 @@ import {
55
56
 
56
57
  const DEFAULT_IDLE_MINUTES = 30
57
58
  const DEFAULT_MIN_CONFIDENCE = 0.8
59
+ /** Terminal sessions older than this (hours since they ended) are not listed
60
+ * as relabel candidates. */
61
+ const DEFAULT_RELABEL_WINDOW_HOURS = 24
62
+ /** Most relabel candidates printed in the report (newest first). */
63
+ export const RELABEL_MAX_LINES = 20
58
64
  /** How many consecutive passes on an unchanged fingerprint before the judge
59
65
  * cache stops re-judging a session. */
60
66
  const DEFAULT_STABLE_VERDICT_PASSES = 2
@@ -104,6 +110,7 @@ export function resolveSettings(input, modelRoles) {
104
110
  return {
105
111
  idleMinutes: Math.floor(num(i.idleMinutes, DEFAULT_IDLE_MINUTES, { min: 1 })),
106
112
  apply: i.apply === true,
113
+ relabelWindowHours: num(i.relabelWindowHours, DEFAULT_RELABEL_WINDOW_HOURS, { min: 0 }),
107
114
  minConfidence: num(i.minConfidence, DEFAULT_MIN_CONFIDENCE, { max: 1 }),
108
115
  judgeModel: explicitOrRole(i.judgeModel, modelRoles, ROLE_JUDGE_SESSION),
109
116
  judge: JUDGE_BACKENDS.includes(i.judge) ? i.judge : "auto",
@@ -635,10 +642,18 @@ export function buildReport(b) {
635
642
  }
636
643
 
637
644
  // Terminal sessions with no outcome (mission item 5).
638
- const relabel = b.steps.relabelQueue ?? []
645
+ const relabel = b.steps.relabelFinal ?? b.steps.relabelQueue ?? []
639
646
  if (relabel.length > 0) {
647
+ const total = b.steps.scan?.counts?.relabelTotal ?? relabel.length
648
+ const byVerdict = new Map()
649
+ for (const r of relabel) byVerdict.set(r.proposedVerdict, (byVerdict.get(r.proposedVerdict) ?? 0) + 1)
650
+ const perLabel = [...byVerdict].map(([v, n]) => `${v}: ${n}`).join(", ")
640
651
  lines.push("", "## Terminal sessions missing an outcome (relabel candidates)")
641
- for (const r of relabel) lines.push(`- ${r.sessionId} → ${r.proposedVerdict} — ${r.reason}`)
652
+ lines.push(`- ${relabel.length} ended in the last ${s.relabelWindowHours}h (${total} without an outcome in all) — ${perLabel}`)
653
+ lines.push(`- evidence: PR numbers come from the session record; worktree/PR state was looked up for the newest ${RELABEL_MAX_LINES} only`)
654
+ for (const r of relabel.slice(0, RELABEL_MAX_LINES)) lines.push(`- ${r.sessionId} → ${r.proposedVerdict} — ${r.reason}`)
655
+ const hidden = total - Math.min(relabel.length, RELABEL_MAX_LINES)
656
+ if (hidden > 0) lines.push(`- … and ${hidden} more (older/omitted)`)
642
657
  }
643
658
 
644
659
  // Verdict memory / cache (mission item 10).
@@ -685,6 +700,8 @@ export function scanLive(liveSessions, settings, nowMs) {
685
700
  const policy = policyOf(settings)
686
701
  const self = settings?.callerSessionId ?? null
687
702
  const idleThreshold = settings?.idleMinutes ?? DEFAULT_IDLE_MINUTES
703
+ const relabelWindowMs = (settings?.relabelWindowHours ?? DEFAULT_RELABEL_WINDOW_HOURS) * 3_600_000
704
+ let relabelTotal = 0
688
705
  const busy = []
689
706
  const idle = []
690
707
  const terminal = []
@@ -724,7 +741,12 @@ export function scanLive(liveSessions, settings, nowMs) {
724
741
  terminal.push({ sessionId: id, origin: s.origin, originClass, label })
725
742
  const cand = terminalRelabelCandidate(s)
726
743
  if (cand.candidate) {
727
- terminalRelabel.push({ sessionId: id, origin: s.origin, originClass, label, proposedVerdict: cand.proposedVerdict, reason: cand.reason })
744
+ relabelTotal++
745
+ const endedAt = s.endedAt ?? s.lastActivityAt ?? s.startedAt
746
+ const endedMs = endedAt ? Date.parse(endedAt) : Number.NaN
747
+ if (Number.isFinite(endedMs) && nowMs - endedMs <= relabelWindowMs) {
748
+ terminalRelabel.push({ sessionId: id, origin: s.origin, originClass, label, proposedVerdict: cand.proposedVerdict, reason: cand.reason, prs: cand.prs ?? [], endedAt, endedMs })
749
+ }
728
750
  }
729
751
  continue
730
752
  }
@@ -735,7 +757,7 @@ export function scanLive(liveSessions, settings, nowMs) {
735
757
  continue
736
758
  }
737
759
  const row = { sessionId: id, origin: s.origin, originClass, label, idleMinutes, lastTurnErroredAt: s.lastTurnErroredAt ?? null }
738
- if (isNeverRan(s)) neverRan.push(row)
760
+ if (isNeverRan(s, { nowMs, idleMinutes: idleThreshold })) neverRan.push(row)
739
761
  if (s.busy === true) {
740
762
  busy.push(row)
741
763
  loopQueue.push({ sessionId: id, originClass, label })
@@ -750,6 +772,7 @@ export function scanLive(liveSessions, settings, nowMs) {
750
772
  idle: idle.length,
751
773
  terminal: terminal.length,
752
774
  terminalRelabel: terminalRelabel.length,
775
+ relabelTotal,
753
776
  neverRan: neverRan.length,
754
777
  excluded: excluded.length,
755
778
  }
@@ -871,9 +894,33 @@ export function buildProposalsStep(scan, loopResults, settings, nowMs) {
871
894
  return { proposals, observed, stalls }
872
895
  }
873
896
 
874
- /** Terminal sessions missing an outcome, as relabel candidates. */
897
+ /** Terminal sessions missing an outcome that ended within the relabel
898
+ * window, newest first (the report caps how many it prints). */
875
899
  export function buildRelabelQueue(scan) {
876
- return (scan?.terminalRelabel ?? []).map(t => ({ ...t }))
900
+ return (scan?.terminalRelabel ?? []).map(t => ({ ...t })).sort((a, b) => b.endedMs - a.endedMs)
901
+ }
902
+
903
+ /** The sessions that get a per-session `session_evidence` lookup: only the
904
+ * newest {@link RELABEL_MAX_LINES} of the (already windowed) relabel queue —
905
+ * the ones the report lists — never every terminal session. */
906
+ export function buildRelabelEvidenceQueue(relabelQueue) {
907
+ return (relabelQueue ?? []).slice(0, RELABEL_MAX_LINES).map(r => ({ sessionId: r.sessionId }))
908
+ }
909
+
910
+ /** One relabel evidence map item: the `session_evidence` answer, id-checked. */
911
+ function foldRelabelEvidence(b) {
912
+ const raw = b.steps.relabelEvidenceOne
913
+ if (!raw || raw.sessionId !== b.item?.sessionId) {
914
+ throw new Error(`session_evidence answered for '${raw?.sessionId}', expected '${b.item?.sessionId}'`)
915
+ }
916
+ return { sessionId: b.item.sessionId, evidence: raw }
917
+ }
918
+
919
+ /** Relabel proposals sharpened by the evidence lookups (a failed lookup
920
+ * leaves its proposal as the list row had it). Order is preserved. */
921
+ export function applyRelabelEvidence(relabelQueue, evidenceResult) {
922
+ const bySession = new Map(settled(evidenceResult).ok.map(r => [r.value?.sessionId, r.value?.evidence]))
923
+ return (relabelQueue ?? []).map(r => (bySession.has(r.sessionId) ? refineRelabel(r, bySession.get(r.sessionId)) : r))
877
924
  }
878
925
 
879
926
  /** The `app_state` events to append for this pass's verdicts. The memory is
@@ -917,6 +964,7 @@ export default {
917
964
  inputs: {
918
965
  idleMinutes: { type: "number", description: `Idle threshold in minutes. Default ${DEFAULT_IDLE_MINUTES}.`, default: DEFAULT_IDLE_MINUTES },
919
966
  apply: { type: "boolean", description: "Close/flag sessions. Default false = dry run (plan + verdicts, no mutation).", default: false },
967
+ relabelWindowHours: { type: "number", description: `Only terminal sessions that ended within this many hours are listed as relabel candidates (at most ${RELABEL_MAX_LINES}, newest first). Default ${DEFAULT_RELABEL_WINDOW_HOURS}.`, default: DEFAULT_RELABEL_WINDOW_HOURS },
920
968
  minConfidence: { type: "number", description: `Judge confidence needed to act. Default ${DEFAULT_MIN_CONFIDENCE}.`, default: DEFAULT_MIN_CONFIDENCE },
921
969
  judge: { type: "string", description: "Judge backend: `auto` (Jev when JEV_API_KEY resolves, else the agent judge), `jev`, or `agent`. A Jev failure always falls back to the agent judge. Default auto.", default: "auto" },
922
970
  jevModel: { type: "string", description: `Jev model. Default ${DEFAULT_JEV_MODEL}.`, default: DEFAULT_JEV_MODEL },
@@ -1008,6 +1056,19 @@ export default {
1008
1056
  // continue, at most one per session per pass, user origins observed only.
1009
1057
  { id: "proposals", kind: "transform", compute: b => buildProposalsStep(b.steps.scan, b.steps.loopResults, b.steps.settings, Date.now()) },
1010
1058
  { id: "relabelQueue", kind: "transform", compute: b => buildRelabelQueue(b.steps.scan) },
1059
+ { id: "relabelEvidenceQueue", kind: "transform", compute: b => buildRelabelEvidenceQueue(b.steps.relabelQueue) },
1060
+ {
1061
+ id: "relabelEvidence",
1062
+ kind: "map",
1063
+ over: "$steps.relabelEvidenceQueue",
1064
+ parallelism: 4,
1065
+ onError: "collect",
1066
+ steps: [
1067
+ { id: "relabelEvidenceOne", kind: "tool", tool: "session_evidence", inputs: { sessionId: "$item.sessionId" } },
1068
+ { id: "relabelEvidenceFold", kind: "transform", compute: foldRelabelEvidence },
1069
+ ],
1070
+ },
1071
+ { id: "relabelFinal", kind: "transform", compute: b => applyRelabelEvidence(b.steps.relabelQueue, b.steps.relabelEvidence) },
1011
1072
  {
1012
1073
  id: "evidence",
1013
1074
  kind: "map",
@@ -1171,7 +1232,7 @@ export default {
1171
1232
  autoApply: "$steps.autoApply",
1172
1233
  judgedApply: "$steps.judgedApply",
1173
1234
  proposals: "$steps.proposals",
1174
- relabel: "$steps.relabelQueue",
1235
+ relabel: "$steps.relabelFinal",
1175
1236
  scan: "$steps.scan",
1176
1237
  },
1177
1238
  }